From 99fcaa4ea5f480b6e8dda05701bcb34ab6c1435c Mon Sep 17 00:00:00 2001 From: Morgan Du Date: Sun, 25 Jul 2021 15:58:16 -0700 Subject: [PATCH] copy notebooks/community from ai-platform-samples (#3) --- ...se_automl_image_classification_batch.ipynb | 1827 +++ ...oml_image_classification_export_edge.ipynb | 1468 +++ ...e_automl_image_classification_online.ipynb | 1587 +++ ...ml_image_classification_online_proxy.ipynb | 1766 +++ ..._automl_image_object_detection_batch.ipynb | 1839 +++ ...l_image_object_detection_export_edge.ipynb | 1471 +++ ...automl_image_object_detection_online.ipynb | 1593 +++ ...case_automl_image_segmentation_batch.ipynb | 1815 +++ ...ase_automl_image_segmentation_online.ipynb | 1574 +++ ..._tabular_binary_classification_batch.ipynb | 1811 +++ ...tabular_binary_classification_online.ipynb | 1687 +++ ..._automl_tabular_classification_batch.ipynb | 1766 +++ ...tabular_classification_batch_explain.ipynb | 1765 +++ ..._tabular_classification_export_cloud.ipynb | 1618 +++ ...automl_tabular_classification_online.ipynb | 1664 +++ ...abular_classification_online_explain.ipynb | 1839 +++ ...ase_automl_tabular_forecasting_batch.ipynb | 1761 +++ ..._automl_tabular_regression_online_bq.ipynb | 1633 +++ ...ase_automl_text_classification_batch.ipynb | 1778 +++ ...se_automl_text_classification_online.ipynb | 1552 +++ ..._automl_text_entity_extraction_batch.ipynb | 1787 +++ ...automl_text_entity_extraction_online.ipynb | 1563 +++ ...ext_multi-label_classification_batch.ipynb | 1779 +++ ...xt_multi-label_classification_online.ipynb | 1552 +++ ...automl_text_sentiment_analysis_batch.ipynb | 1777 +++ ...utoml_text_sentiment_analysis_online.ipynb | 1553 +++ ...utoml_video_action_recognition_batch.ipynb | 1771 +++ ...se_automl_video_classification_batch.ipynb | 1764 +++ ...e_automl_video_object_tracking_batch.ipynb | 1772 +++ ...se_custom_image_classification_batch.ipynb | 2257 ++++ ...m_image_classification_batch_explain.ipynb | 2465 ++++ ...e_custom_image_classification_online.ipynb | 2192 ++++ ...age_classification_online_ab_testing.ipynb | 2444 ++++ ...mage_classification_online_container.ipynb | 2294 ++++ ..._image_classification_online_explain.ipynb | 2514 ++++ ...ge_classification_online_exported_ds.ipynb | 2795 +++++ ...image_classification_online_pipeline.ipynb | 2261 ++++ ...stom_image_classification_online_raw.ipynb | 2179 ++++ ...mage_classification_online_tfserving.ipynb | 2269 ++++ ...m_image_super_resolution_online_post.ipynb | 2188 ++++ ...ar_classification_online_exported_ds.ipynb | 2339 ++++ ...case_custom_tabular_regression_batch.ipynb | 2143 ++++ ...tom_tabular_regression_batch_explain.ipynb | 2342 ++++ ...ase_custom_tabular_regression_online.ipynb | 2121 ++++ ..._tabular_regression_online_container.ipynb | 2215 ++++ ...om_tabular_regression_online_explain.ipynb | 2498 ++++ ...m_tabular_regression_online_pipeline.ipynb | 2187 ++++ ..._tabular_regression_online_tfserving.ipynb | 2190 ++++ ...tom_text_binary_classification_batch.ipynb | 2112 ++++ ...om_text_binary_classification_online.ipynb | 2090 ++++ ...nary_classification_online_container.ipynb | 2192 ++++ ...inary_classification_online_pipeline.ipynb | 2159 ++++ ...nary_classification_online_tfserving.ipynb | 2169 ++++ ...xt_classification_online_exported_ds.ipynb | 1305 ++ ...parmeter_tuning_image_classification.ipynb | 2081 ++++ ...erparmeter_tuning_tabular_regression.ipynb | 2104 ++++ ...er_tuning_text_binary_classification.ipynb | 2079 ++++ ...se_local_image_classification_online.ipynb | 1737 +++ ...case_local_tabular_regression_online.ipynb | 1658 +++ ...al_text_binary_classification_online.ipynb | 1635 +++ ...se_tfhub_image_classification_online.ipynb | 1490 +++ .../distributed-hyperparameter-tuning.ipynb | 889 ++ .../hyperparameter_tuning/tuning_results.png | Bin 0 -> 272299 bytes .../matching_engine_for_indexing.ipynb | 1770 +++ notebooks/community/migration/README.md | 0 ...y AutoML Vision Image Classification.ipynb | 3595 ++++++ ...d AutoML Vision Image Classification.ipynb | 3274 +++++ ... Training Prebuilt Container SKLearn.ipynb | 1942 +++ ... Training Prebuilt Container SKLearn.ipynb | 2649 ++++ ... Tuning Training Job with TensorFlow.ipynb | 1398 +++ ... Tuning Training Job with TensorFlow.ipynb | 1373 +++ .../UJ13 legacy Data Labeling task.ipynb | 1485 +++ .../UJ13 unified Data Labeling task.ipynb | 1463 +++ ...y AutoML Vision Video Classification.ipynb | 1554 +++ ...d AutoML Vision Video Classification.ipynb | 1996 +++ ... AutoML Vision Video Object Tracking.ipynb | 1625 +++ ... AutoML Vision Video Object Tracking.ipynb | 1902 +++ ...Training Prebuilt Container TF Keras.ipynb | 2065 ++++ ...Training Prebuilt Container TF Keras.ipynb | 2253 ++++ ...Training Custom Container (TF Keras).ipynb | 2001 +++ ...Training Custom Container (TF Keras).ipynb | 2333 ++++ .../UJ4 legacy AutoML Tables Regression.ipynb | 10233 ++++++++++++++++ ...UJ4 unified AutoML Tables Regression.ipynb | 2482 ++++ ...utoML Vision Images Object Detection.ipynb | 1839 +++ ...d AutoML Vision Video Classification.ipynb | 2480 ++++ ...Natural Language Text Classification.ipynb | 1691 +++ ...Natural Language Text Classification.ipynb | 2535 ++++ ...ural Language Text Entity Extraction.ipynb | 1633 +++ ...ural Language Text Entity Extraction.ipynb | 2417 ++++ ...l Language - Text Sentiment Analysis.ipynb | 1634 +++ ...l Language - Text Sentiment Analysis.ipynb | 2355 ++++ ... Training Prebuilt Container XGBoost.ipynb | 1542 +++ ... Training Prebuilt Container XGBoost.ipynb | 2063 ++++ ...c-parameter-tracking-for-custom-jobs.ipynb | 1038 ++ ...-tracking-for-locally-trained-models.ipynb | 877 ++ ...L_Forecasting_Model_Training_Example.ipynb | 547 + ...ge_Classification_Training_with_CMEK.ipynb | 522 + .../SDK_AutoML_Text_Extraction_Training.ipynb | 313 + .../SDK_AutoML_Video_Action_Recognition.ipynb | 501 + .../sdk/SDK_AutoML_Video_Classification.ipynb | 496 + ...K_BigQuery_Custom_Container_Training.ipynb | 521 + .../sdk/SDK_Custom_Container_Prediction.ipynb | 1118 ++ ...aining_AutoML_Tabular_Model_Training.ipynb | 452 + ...Dataset_Tensorflow_Serving_Container.ipynb | 1259 ++ ...raining_with_Unmanaged_Image_Dataset.ipynb | 396 + ...K_End_to_End_Tabular_Custom_Training.ipynb | 335 + .../SDK_Explainable_AI_Custom_Tabular.ipynb | 608 + ...r_Custom_Model_Training_asynchronous.ipynb | 357 + ...dk_automl_image_classification_batch.ipynb | 1180 ++ ...k_automl_image_classification_online.ipynb | 1069 ++ ..._automl_image_object_detection_batch.ipynb | 1189 ++ ...automl_image_object_detection_online.ipynb | 1075 ++ ..._tabular_binary_classification_batch.ipynb | 1132 ++ ...tabular_binary_classification_online.ipynb | 1048 ++ ...sdk_automl_text_classification_batch.ipynb | 1170 ++ ...dk_automl_text_classification_online.ipynb | 1046 ++ ...automl_text_sentiment_analysis_batch.ipynb | 1172 ++ ...utoml_text_sentiment_analysis_online.ipynb | 1041 ++ 118 files changed, 206769 insertions(+) create mode 100644 notebooks/community/gapic/automl/showcase_automl_image_classification_batch.ipynb create mode 100644 notebooks/community/gapic/automl/showcase_automl_image_classification_export_edge.ipynb create mode 100644 notebooks/community/gapic/automl/showcase_automl_image_classification_online.ipynb create mode 100644 notebooks/community/gapic/automl/showcase_automl_image_classification_online_proxy.ipynb create mode 100644 notebooks/community/gapic/automl/showcase_automl_image_object_detection_batch.ipynb create mode 100644 notebooks/community/gapic/automl/showcase_automl_image_object_detection_export_edge.ipynb create mode 100644 notebooks/community/gapic/automl/showcase_automl_image_object_detection_online.ipynb create mode 100644 notebooks/community/gapic/automl/showcase_automl_image_segmentation_batch.ipynb create mode 100644 notebooks/community/gapic/automl/showcase_automl_image_segmentation_online.ipynb create mode 100644 notebooks/community/gapic/automl/showcase_automl_tabular_binary_classification_batch.ipynb create mode 100644 notebooks/community/gapic/automl/showcase_automl_tabular_binary_classification_online.ipynb create mode 100644 notebooks/community/gapic/automl/showcase_automl_tabular_classification_batch.ipynb create mode 100644 notebooks/community/gapic/automl/showcase_automl_tabular_classification_batch_explain.ipynb create mode 100644 notebooks/community/gapic/automl/showcase_automl_tabular_classification_export_cloud.ipynb create mode 100644 notebooks/community/gapic/automl/showcase_automl_tabular_classification_online.ipynb create mode 100644 notebooks/community/gapic/automl/showcase_automl_tabular_classification_online_explain.ipynb create mode 100644 notebooks/community/gapic/automl/showcase_automl_tabular_forecasting_batch.ipynb create mode 100644 notebooks/community/gapic/automl/showcase_automl_tabular_regression_online_bq.ipynb create mode 100644 notebooks/community/gapic/automl/showcase_automl_text_classification_batch.ipynb create mode 100644 notebooks/community/gapic/automl/showcase_automl_text_classification_online.ipynb create mode 100644 notebooks/community/gapic/automl/showcase_automl_text_entity_extraction_batch.ipynb create mode 100644 notebooks/community/gapic/automl/showcase_automl_text_entity_extraction_online.ipynb create mode 100644 notebooks/community/gapic/automl/showcase_automl_text_multi-label_classification_batch.ipynb create mode 100644 notebooks/community/gapic/automl/showcase_automl_text_multi-label_classification_online.ipynb create mode 100644 notebooks/community/gapic/automl/showcase_automl_text_sentiment_analysis_batch.ipynb create mode 100644 notebooks/community/gapic/automl/showcase_automl_text_sentiment_analysis_online.ipynb create mode 100644 notebooks/community/gapic/automl/showcase_automl_video_action_recognition_batch.ipynb create mode 100644 notebooks/community/gapic/automl/showcase_automl_video_classification_batch.ipynb create mode 100644 notebooks/community/gapic/automl/showcase_automl_video_object_tracking_batch.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_custom_image_classification_batch.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_custom_image_classification_batch_explain.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_custom_image_classification_online.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_custom_image_classification_online_ab_testing.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_custom_image_classification_online_container.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_custom_image_classification_online_explain.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_custom_image_classification_online_exported_ds.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_custom_image_classification_online_pipeline.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_custom_image_classification_online_raw.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_custom_image_classification_online_tfserving.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_custom_image_super_resolution_online_post.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_custom_tabular_classification_online_exported_ds.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_custom_tabular_regression_batch.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_custom_tabular_regression_batch_explain.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_custom_tabular_regression_online.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_custom_tabular_regression_online_container.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_custom_tabular_regression_online_explain.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_custom_tabular_regression_online_pipeline.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_custom_tabular_regression_online_tfserving.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_custom_text_binary_classification_batch.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_custom_text_binary_classification_online.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_custom_text_binary_classification_online_container.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_custom_text_binary_classification_online_pipeline.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_custom_text_binary_classification_online_tfserving.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_custom_text_classification_online_exported_ds.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_hyperparmeter_tuning_image_classification.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_hyperparmeter_tuning_tabular_regression.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_hyperparmeter_tuning_text_binary_classification.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_local_image_classification_online.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_local_tabular_regression_online.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_local_text_binary_classification_online.ipynb create mode 100644 notebooks/community/gapic/custom/showcase_tfhub_image_classification_online.ipynb create mode 100644 notebooks/community/hyperparameter_tuning/distributed-hyperparameter-tuning.ipynb create mode 100644 notebooks/community/hyperparameter_tuning/tuning_results.png create mode 100644 notebooks/community/matching_engine/matching_engine_for_indexing.ipynb create mode 100644 notebooks/community/migration/README.md create mode 100644 notebooks/community/migration/UJ1 legacy AutoML Vision Image Classification.ipynb create mode 100644 notebooks/community/migration/UJ1 unified AutoML Vision Image Classification.ipynb create mode 100644 notebooks/community/migration/UJ10 legacy Custom Training Prebuilt Container SKLearn.ipynb create mode 100644 notebooks/community/migration/UJ10 unified Custom Training Prebuilt Container SKLearn.ipynb create mode 100644 notebooks/community/migration/UJ11 legacy HyperParameter Tuning Training Job with TensorFlow.ipynb create mode 100644 notebooks/community/migration/UJ11 unified HyperParameter Tuning Training Job with TensorFlow.ipynb create mode 100644 notebooks/community/migration/UJ13 legacy Data Labeling task.ipynb create mode 100644 notebooks/community/migration/UJ13 unified Data Labeling task.ipynb create mode 100644 notebooks/community/migration/UJ14 legacy AutoML Vision Video Classification.ipynb create mode 100644 notebooks/community/migration/UJ14 unified AutoML Vision Video Classification.ipynb create mode 100644 notebooks/community/migration/UJ15 legacy AutoML Vision Video Object Tracking.ipynb create mode 100644 notebooks/community/migration/UJ15 unified AutoML Vision Video Object Tracking.ipynb create mode 100644 notebooks/community/migration/UJ2,12 legacy Custom Training Prebuilt Container TF Keras.ipynb create mode 100644 notebooks/community/migration/UJ2,12 unified Custom Training Prebuilt Container TF Keras.ipynb create mode 100644 notebooks/community/migration/UJ3 legacy Custom Training Custom Container (TF Keras).ipynb create mode 100644 notebooks/community/migration/UJ3 unified Custom Training Custom Container (TF Keras).ipynb create mode 100644 notebooks/community/migration/UJ4 legacy AutoML Tables Regression.ipynb create mode 100644 notebooks/community/migration/UJ4 unified AutoML Tables Regression.ipynb create mode 100644 notebooks/community/migration/UJ5 legacy AutoML Vision Images Object Detection.ipynb create mode 100644 notebooks/community/migration/UJ5 unified AutoML Vision Video Classification.ipynb create mode 100644 notebooks/community/migration/UJ6 legacy AutoML Natural Language Text Classification.ipynb create mode 100644 notebooks/community/migration/UJ6 unified AutoML Natural Language Text Classification.ipynb create mode 100644 notebooks/community/migration/UJ7 legacy AutoML Natural Language Text Entity Extraction.ipynb create mode 100644 notebooks/community/migration/UJ7 unified AutoML Natural Language Text Entity Extraction.ipynb create mode 100644 notebooks/community/migration/UJ8 legacy AutoML Natural Language - Text Sentiment Analysis.ipynb create mode 100644 notebooks/community/migration/UJ8 unified AutoML Natural Language - Text Sentiment Analysis.ipynb create mode 100644 notebooks/community/migration/UJ9 legacy Custom Training Prebuilt Container XGBoost.ipynb create mode 100644 notebooks/community/migration/UJ9 unified Custom Training Prebuilt Container XGBoost.ipynb create mode 100644 notebooks/community/ml_metadata/sdk-metric-parameter-tracking-for-custom-jobs.ipynb create mode 100644 notebooks/community/ml_metadata/sdk-metric-parameter-tracking-for-locally-trained-models.ipynb create mode 100644 notebooks/community/sdk/SDK_AutoML_Forecasting_Model_Training_Example.ipynb create mode 100644 notebooks/community/sdk/SDK_AutoML_Image_Classification_Training_with_CMEK.ipynb create mode 100644 notebooks/community/sdk/SDK_AutoML_Text_Extraction_Training.ipynb create mode 100644 notebooks/community/sdk/SDK_AutoML_Video_Action_Recognition.ipynb create mode 100644 notebooks/community/sdk/SDK_AutoML_Video_Classification.ipynb create mode 100644 notebooks/community/sdk/SDK_BigQuery_Custom_Container_Training.ipynb create mode 100644 notebooks/community/sdk/SDK_Custom_Container_Prediction.ipynb create mode 100644 notebooks/community/sdk/SDK_Custom_Model_Training_AutoML_Tabular_Model_Training.ipynb create mode 100644 notebooks/community/sdk/SDK_Custom_Training_Python_Package_Managed_Text_Dataset_Tensorflow_Serving_Container.ipynb create mode 100644 notebooks/community/sdk/SDK_Custom_Training_with_Unmanaged_Image_Dataset.ipynb create mode 100644 notebooks/community/sdk/SDK_End_to_End_Tabular_Custom_Training.ipynb create mode 100644 notebooks/community/sdk/SDK_Explainable_AI_Custom_Tabular.ipynb create mode 100644 notebooks/community/sdk/SDK_Tabular_Custom_Model_Training_asynchronous.ipynb create mode 100644 notebooks/community/sdk/sdk_automl_image_classification_batch.ipynb create mode 100644 notebooks/community/sdk/sdk_automl_image_classification_online.ipynb create mode 100644 notebooks/community/sdk/sdk_automl_image_object_detection_batch.ipynb create mode 100644 notebooks/community/sdk/sdk_automl_image_object_detection_online.ipynb create mode 100644 notebooks/community/sdk/sdk_automl_tabular_binary_classification_batch.ipynb create mode 100644 notebooks/community/sdk/sdk_automl_tabular_binary_classification_online.ipynb create mode 100644 notebooks/community/sdk/sdk_automl_text_classification_batch.ipynb create mode 100644 notebooks/community/sdk/sdk_automl_text_classification_online.ipynb create mode 100644 notebooks/community/sdk/sdk_automl_text_sentiment_analysis_batch.ipynb create mode 100644 notebooks/community/sdk/sdk_automl_text_sentiment_analysis_online.ipynb diff --git a/notebooks/community/gapic/automl/showcase_automl_image_classification_batch.ipynb b/notebooks/community/gapic/automl/showcase_automl_image_classification_batch.ipynb new file mode 100644 index 000000000..1c6552e15 --- /dev/null +++ b/notebooks/community/gapic/automl/showcase_automl_image_classification_batch.ipynb @@ -0,0 +1,1827 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: AutoML image classification model for batch prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to create image classification models and do batch prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:flowers,icn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Flowers dataset](https://www.tensorflow.org/datasets/catalog/tf_flowers) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts the type of flower an image is from a class of five flowers: daisy, dandelion, rose, sunflower, or tulip." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,batch_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML image classification model from a Python script, and then do a batch prediction using the Vertex client library. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Make a batch prediction.\n", + "\n", + "There is one key difference between using batch prediction and using online prediction:\n", + "\n", + "* Prediction Service: Does an on-demand prediction for the entire set of instances (i.e., one or more data items) and returns the results in real-time.\n", + "\n", + "* Batch Prediction Service: Does a queued (batch) prediction for the entire set of instances in the background and stores the results in a Cloud Storage bucket when ready." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Set constants unique to AutoML datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the `Dataset` resource service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the `Dataset` resource service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:icn" + }, + "outputs": [], + "source": [ + "# Image Dataset type\n", + "DATA_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/metadata/image_1.0.0.yaml\"\n", + "# Image Labeling type\n", + "LABEL_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/ioformat/image_classification_single_label_io_format_1.0.0.yaml\"\n", + "# Image Training task\n", + "TRAINING_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_image_classification_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:prediction" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for prediction.\n", + "\n", + "Set the variable `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:automl" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "For AutoML batch prediction, the container image for the serving binary is pre-determined by the Vertex prediction service. More specifically, the service will pick the appropriate container for the model depending on the hardware accelerator you selected." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for prediction.\n", + "\n", + "- Set the variable `DEPLOY_COMPUTE` to configure the compute resources for the VM you will use for prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML image classification model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:automl,batch_prediction" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training.\n", + "- Job Service for batch prediction and custom training." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:automl,batch_prediction" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"job\"] = create_job_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "\n", + "def create_dataset(name, schema, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " dataset = aip.Dataset(\n", + " display_name=name, metadata_schema_uri=schema, labels=labels\n", + " )\n", + "\n", + " operation = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"flowers-\" + TIMESTAMP, DATA_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:image,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for images has some requirements for your data:\n", + "\n", + "- Images must be stored in a Cloud Storage bucket.\n", + "- Each image file must be in an image format (PNG, JPEG, BMP, ...).\n", + "- There must be an index file stored in your Cloud Storage bucket that contains the path and label for each image.\n", + "- The index file must be either CSV or JSONL." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:icn,u_dataset,csv" + }, + "source": [ + "#### CSV\n", + "\n", + "For image classification, the CSV index file has the requirements:\n", + "\n", + "- No heading.\n", + "- First column is the Cloud Storage path to the image.\n", + "- Second column is the label." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:flowers,csv,icn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = (\n", + " \"gs://cloud-samples-data/vision/automl_classification/flowers/all_data_v2.csv\"\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:csv" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Flowers dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:csv" + }, + "outputs": [], + "source": [ + "if \"IMPORT_FILES\" in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_data" + }, + "source": [ + "### Import data\n", + "\n", + "Now, import the data into your Vertex Dataset resource. Use this helper function `import_data` to import the data. The function does the following:\n", + "\n", + "- Uses the `Dataset` client.\n", + "- Calls the client method `import_data`, with the following parameters:\n", + " - `name`: The human readable name you give to the `Dataset` resource (e.g., flowers).\n", + " - `import_configs`: The import configuration.\n", + "\n", + "- `import_configs`: A Python list containing a dictionary, with the key/value entries:\n", + " - `gcs_sources`: A list of URIs to the paths of the one or more index files.\n", + " - `import_schema_uri`: The schema identifying the labeling type.\n", + "\n", + "The `import_data()` method returns a long running `operation` object. This will take a few minutes to complete. If you are in a live tutorial, this would be a good time to ask questions, or take a personal break." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_data" + }, + "outputs": [], + "source": [ + "def import_data(dataset, gcs_sources, schema):\n", + " config = [{\"gcs_source\": {\"uris\": gcs_sources}, \"import_schema_uri\": schema}]\n", + " print(\"dataset:\", dataset_id)\n", + " start_time = time.time()\n", + " try:\n", + " operation = clients[\"dataset\"].import_data(\n", + " name=dataset_id, import_configs=config\n", + " )\n", + " print(\"Long running operation:\", operation.operation.name)\n", + "\n", + " result = operation.result()\n", + " print(\"result:\", result)\n", + " print(\"time:\", int(time.time() - start_time), \"secs\")\n", + " print(\"error:\", operation.exception())\n", + " print(\"meta :\", operation.metadata)\n", + " print(\n", + " \"after: running:\",\n", + " operation.running(),\n", + " \"done:\",\n", + " operation.done(),\n", + " \"cancelled:\",\n", + " operation.cancelled(),\n", + " )\n", + "\n", + " return operation\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "import_data(dataset_id, [IMPORT_FILE], LABEL_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML image classification model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_pipeline:automl" + }, + "source": [ + "### Create a training pipeline\n", + "\n", + "You may ask, what do we use a pipeline for? You typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `pipeline_name`: A human readable name for the pipeline job.\n", + "- `model_name`: A human readable name for the model.\n", + "- `dataset`: The Vertex fully qualified dataset identifier.\n", + "- `schema`: The dataset labeling (annotation) training schema.\n", + "- `task`: A dictionary describing the requirements for the training job.\n", + "\n", + "The helper function calls the `Pipeline` client service'smethod `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "Let's look now deeper into the *minimal* requirements for constructing a `training_pipeline` specification:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The dataset labeling (annotation) training schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A human readable name for the model.\n", + "- `input_data_config`: The dataset specification.\n", + " - `dataset_id`: The Vertex dataset identifier only (non-fully qualified) -- this is the last part of the fully-qualified identifier.\n", + " - `fraction_split`: If specified, the percentages of the dataset to use for training, test and validation. Otherwise, the percentages are automatically selected by AutoML." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_pipeline:automl" + }, + "outputs": [], + "source": [ + "def create_pipeline(pipeline_name, model_name, dataset, schema, task):\n", + "\n", + " dataset_id = dataset.split(\"/\")[-1]\n", + "\n", + " input_config = {\n", + " \"dataset_id\": dataset_id,\n", + " \"fraction_split\": {\n", + " \"training_fraction\": 0.8,\n", + " \"validation_fraction\": 0.1,\n", + " \"test_fraction\": 0.1,\n", + " },\n", + " }\n", + "\n", + " training_pipeline = {\n", + " \"display_name\": pipeline_name,\n", + " \"training_task_definition\": schema,\n", + " \"training_task_inputs\": task,\n", + " \"input_data_config\": input_config,\n", + " \"model_to_upload\": {\"display_name\": model_name},\n", + " }\n", + "\n", + " try:\n", + " pipeline = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " )\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "task_requirements:automl,icn" + }, + "source": [ + "### Construct the task requirements\n", + "\n", + "Next, construct the task requirements. Unlike other parameters which take a Python (JSON-like) dictionary, the `task` field takes a Google protobuf Struct, which is very similar to a Python dictionary. Use the `json_format.ParseDict` method for the conversion.\n", + "\n", + "The minimal fields we need to specify are:\n", + "\n", + "- `multi_label`: Whether True/False this is a multi-label (vs single) classification.\n", + "- `budget_milli_node_hours`: The maximum time to budget (billed) for training the model, where 1000 = 1 hour. For image classification, the budget must be a minimum of 8 hours.\n", + "- `model_type`: The type of deployed model:\n", + " - `CLOUD`: For deploying to Google Cloud.\n", + " - `MOBILE_TF_LOW_LATENCY_1`: For deploying to the edge and optimizing for latency (response time).\n", + " - `MOBILE_TF_HIGH_ACCURACY_1`: For deploying to the edge and optimizing for accuracy.\n", + " - `MOBILE_TF_VERSATILE_1`: For deploying to the edge and optimizing for a trade off between latency and accuracy.\n", + "- `disable_early_stopping`: Whether True/False to let AutoML use its judgement to stop training early or train for the entire budget.\n", + "\n", + "Finally, create the pipeline by calling the helper function `create_pipeline`, which returns an instance of a training pipeline object." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_requirements:automl,icn" + }, + "outputs": [], + "source": [ + "PIPE_NAME = \"flowers_pipe-\" + TIMESTAMP\n", + "MODEL_NAME = \"flowers_model-\" + TIMESTAMP\n", + "\n", + "task = json_format.ParseDict(\n", + " {\n", + " \"multi_label\": False,\n", + " \"budget_milli_node_hours\": 8000,\n", + " \"model_type\": \"CLOUD\",\n", + " \"disable_early_stopping\": False,\n", + " },\n", + " Value(),\n", + ")\n", + "\n", + "response = create_pipeline(PIPE_NAME, MODEL_NAME, dataset_id, TRAINING_SCHEMA, task)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "model_information" + }, + "source": [ + "## Model information\n", + "\n", + "Now that your model is trained, you can get some information on your model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:automl" + }, + "source": [ + "## Evaluate the Model resource\n", + "\n", + "Now find out how good the model service believes your model is. As part of training, some portion of the dataset was set aside as the test (holdout) data, which is used by the pipeline service to evaluate the model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "list_model_evaluations:automl,icn" + }, + "source": [ + "### List evaluations for all slices\n", + "\n", + "Use this helper function `list_model_evaluations`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified model identifier for the `Model` resource.\n", + "\n", + "This helper function uses the model client service's `list_model_evaluations` method, which takes the same parameter. The response object from the call is a list, where each element is an evaluation metric.\n", + "\n", + "For each evaluation (you probably only have one) we then print all the key names for each metric in the evaluation, and for a small set (`logLoss` and `auPrc`) you will print the result." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "list_model_evaluations:automl,icn" + }, + "outputs": [], + "source": [ + "def list_model_evaluations(name):\n", + " response = clients[\"model\"].list_model_evaluations(parent=name)\n", + " for evaluation in response:\n", + " print(\"model_evaluation\")\n", + " print(\" name:\", evaluation.name)\n", + " print(\" metrics_schema_uri:\", evaluation.metrics_schema_uri)\n", + " metrics = json_format.MessageToDict(evaluation._pb.metrics)\n", + " for metric in metrics.keys():\n", + " print(metric)\n", + " print(\"logloss\", metrics[\"logLoss\"])\n", + " print(\"auPrc\", metrics[\"auPrc\"])\n", + "\n", + " return evaluation.name\n", + "\n", + "\n", + "last_evaluation = list_model_evaluations(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy:batch_prediction" + }, + "source": [ + "## Model deployment for batch prediction\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created for batch prediction. This differs from deploying a `Model` resource for on-demand prediction.\n", + "\n", + "For online prediction, you:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource.\n", + "\n", + "3. Make online prediction requests to the `Endpoint` resource.\n", + "\n", + "For batch-prediction, you:\n", + "\n", + "1. Create a batch prediction job.\n", + "\n", + "2. The job service will provision resources for the batch prediction request.\n", + "\n", + "3. The results of the batch prediction request are returned to the caller.\n", + "\n", + "4. The job service will unprovision the resoures for the batch prediction request." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a batch prediction request\n", + "\n", + "Now do a batch prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_items:batch_prediction" + }, + "source": [ + "### Get test item(s)\n", + "\n", + "Now do a batch prediction to your Vertex model. You will use arbitrary examples out of the dataset as a test items. Don't be concerned that the examples were likely used in training the model -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_items:automl,icn,csv" + }, + "outputs": [], + "source": [ + "test_items = !gsutil cat $IMPORT_FILE | head -n2\n", + "if len(str(test_items[0]).split(\",\")) == 3:\n", + " _, test_item_1, test_label_1 = str(test_items[0]).split(\",\")\n", + " _, test_item_2, test_label_2 = str(test_items[1]).split(\",\")\n", + "else:\n", + " test_item_1, test_label_1 = str(test_items[0]).split(\",\")\n", + " test_item_2, test_label_2 = str(test_items[1]).split(\",\")\n", + "\n", + "print(test_item_1, test_label_1)\n", + "print(test_item_2, test_label_2)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "copy_test_items:batch_prediction" + }, + "source": [ + "### Copy test item(s)\n", + "\n", + "For the batch prediction, you will copy the test items over to your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copy_test_items:batch_prediction" + }, + "outputs": [], + "source": [ + "file_1 = test_item_1.split(\"/\")[-1]\n", + "file_2 = test_item_2.split(\"/\")[-1]\n", + "\n", + "! gsutil cp $test_item_1 $BUCKET_NAME/$file_1\n", + "! gsutil cp $test_item_2 $BUCKET_NAME/$file_2\n", + "\n", + "test_item_1 = BUCKET_NAME + \"/\" + file_1\n", + "test_item_2 = BUCKET_NAME + \"/\" + file_2" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:automl,image" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "Now make a batch input file, which you will store in your local Cloud Storage bucket. The batch input file can be either CSV or JSONL. You will use JSONL in this tutorial. For JSONL file, you make one dictionary entry per line for each data item (instance). The dictionary contains the key/value pairs:\n", + "\n", + "- `content`: The Cloud Storage path to the image.\n", + "- `mime_type`: The content type. In our example, it is an `jpeg` file.\n", + "\n", + "For example:\n", + "\n", + " {'content': '[your-bucket]/file1.jpg', 'mime_type': 'jpeg'}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_file:automl,image" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "import tensorflow as tf\n", + "\n", + "gcs_input_uri = BUCKET_NAME + \"/test.jsonl\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " data = {\"content\": test_item_1, \"mime_type\": \"image/jpeg\"}\n", + " f.write(json.dumps(data) + \"\\n\")\n", + " data = {\"content\": test_item_2, \"mime_type\": \"image/jpeg\"}\n", + " f.write(json.dumps(data) + \"\\n\")\n", + "\n", + "print(gcs_input_uri)\n", + "! gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your batch prediction requests:\n", + "\n", + "- Single Instance: The batch prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The batch prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and batch prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The batch prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_request:automl,icn" + }, + "source": [ + "### Make batch prediction request\n", + "\n", + "Now that your batch of two test items is ready, let's do the batch request. Use this helper function `create_batch_prediction_job`, with the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the prediction job.\n", + "- `model_name`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `gcs_source_uri`: The Cloud Storage path to the input file -- which you created above.\n", + "- `gcs_destination_output_uri_prefix`: The Cloud Storage path that the service will write the predictions to.\n", + "- `parameters`: Additional filtering parameters for serving prediction results.\n", + "\n", + "The helper function calls the job client service's `create_batch_prediction_job` metho, with the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for Dataset, Model and Pipeline resources.\n", + "- `batch_prediction_job`: The specification for the batch prediction job.\n", + "\n", + "Let's now dive into the specification for the `batch_prediction_job`:\n", + "\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `dedicated_resources`: The compute resources to provision for the batch prediction job.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `starting_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "- `model_parameters`: Additional filtering parameters for serving prediction results.\n", + " - `confidence_threshold`: The threshold for returning predictions. Must be between 0 and 1.\n", + " - `max_predictions`: The maximum number of predictions to return per classification, sorted by confidence.\n", + "- `input_config`: The input source and format type for the instances to predict.\n", + " - `instances_format`: The format of the batch prediction request file: `jsonl` only supported.\n", + " - `gcs_source`: A list of one or more Cloud Storage paths to your batch prediction requests.\n", + "- `output_config`: The output destination and format for the predictions.\n", + " - `prediction_format`: The format of the batch prediction response file: `jsonl` only supported.\n", + " - `gcs_destination`: The output destination for the predictions.\n", + "\n", + "You might ask, how does confidence_threshold affect the model accuracy? The threshold won't change the accuracy. What it changes is recall and precision.\n", + "\n", + " - Precision: The higher the precision the more likely what is predicted is the correct prediction, but return fewer predictions. Increasing the confidence threshold increases precision.\n", + " - Recall: The higher the recall the more likely a correct prediction is returned in the result, but return more prediction with incorrect prediction. Decreasing the confidence threshold increases recall.\n", + "\n", + "In this example, you will predict for precision. You set the confidence threshold to 0.5 and the maximum number of predictions for a classification to two. Since, all the confidence values across the classes must add up to one, there are only two possible outcomes:\n", + "\n", + " 1. There is a tie, both 0.5, and returns two predictions.\n", + " 2. One value is above 0.5 and the rest are below 0.5, and returns one prediction.\n", + "\n", + "This call is an asychronous operation. You will print from the response object a few select fields, including:\n", + "\n", + "- `name`: The Vertex fully qualified identifier assigned to the batch prediction job.\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the Model resource.\n", + "- `generate_explanations`: Whether True/False explanations were provided with the predictions (explainability).\n", + "- `state`: The state of the prediction job (pending, running, etc).\n", + "\n", + "Since this call will take a few moments to execute, you will likely get `JobState.JOB_STATE_PENDING` for `state`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_request:automl,icn" + }, + "outputs": [], + "source": [ + "BATCH_MODEL = \"flowers_batch-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_batch_prediction_job(\n", + " display_name,\n", + " model_name,\n", + " gcs_source_uri,\n", + " gcs_destination_output_uri_prefix,\n", + " parameters=None,\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " batch_prediction_job = {\n", + " \"display_name\": display_name,\n", + " # Format: 'projects/{project}/locations/{location}/models/{model_id}'\n", + " \"model\": model_name,\n", + " \"model_parameters\": json_format.ParseDict(parameters, Value()),\n", + " \"input_config\": {\n", + " \"instances_format\": IN_FORMAT,\n", + " \"gcs_source\": {\"uris\": [gcs_source_uri]},\n", + " },\n", + " \"output_config\": {\n", + " \"predictions_format\": OUT_FORMAT,\n", + " \"gcs_destination\": {\"output_uri_prefix\": gcs_destination_output_uri_prefix},\n", + " },\n", + " \"dedicated_resources\": {\n", + " \"machine_spec\": machine_spec,\n", + " \"starting_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " },\n", + " }\n", + " response = clients[\"job\"].create_batch_prediction_job(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + " )\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try:\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", response.labels)\n", + " return response\n", + "\n", + "\n", + "IN_FORMAT = \"jsonl\"\n", + "OUT_FORMAT = \"jsonl\" # [jsonl]\n", + "\n", + "response = create_batch_prediction_job(\n", + " BATCH_MODEL,\n", + " model_to_deploy_id,\n", + " gcs_input_uri,\n", + " BUCKET_NAME,\n", + " {\"confidenceThreshold\": 0.5, \"maxPredictions\": 2},\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batch_job_id:response" + }, + "source": [ + "Now get the unique identifier for the batch prediction job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batch_job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the batch job\n", + "batch_job_id = response.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split(\"/\")[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_batch_prediction_job" + }, + "source": [ + "### Get information on a batch prediction job\n", + "\n", + "Use this helper function `get_batch_prediction_job`, with the following paramter:\n", + "\n", + "- `job_name`: The Vertex fully qualified identifier for the batch prediction job.\n", + "\n", + "The helper function calls the job client service's `get_batch_prediction_job` method, with the following paramter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the batch prediction job. In this tutorial, you will pass it the Vertex fully qualified identifier for your batch prediction job -- `batch_job_id`\n", + "\n", + "The helper function will return the Cloud Storage path to where the predictions are stored -- `gcs_destination`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_batch_prediction_job" + }, + "outputs": [], + "source": [ + "def get_batch_prediction_job(job_name, silent=False):\n", + " response = clients[\"job\"].get_batch_prediction_job(name=job_name)\n", + " if silent:\n", + " return response.output_config.gcs_destination.output_uri_prefix, response.state\n", + "\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try: # not all data types support explanations\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" error:\", response.error)\n", + " gcs_destination = response.output_config.gcs_destination\n", + " print(\" gcs_destination\")\n", + " print(\" output_uri_prefix:\", gcs_destination.output_uri_prefix)\n", + " return gcs_destination.output_uri_prefix, response.state\n", + "\n", + "\n", + "predictions, state = get_batch_prediction_job(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_the_predictions:automl,icn" + }, + "source": [ + "### Get the predictions\n", + "\n", + "When the batch prediction is done processing, the job state will be `JOB_STATE_SUCCEEDED`.\n", + "\n", + "Finally you view the predictions stored at the Cloud Storage path you set as output. The predictions will be in a JSONL format, which you indicated at the time you made the batch prediction job, under a subfolder starting with the name `prediction`, and under that folder will be a file called `predictions*.jsonl`.\n", + "\n", + "Now display (cat) the contents. You will see multiple JSON objects, one for each prediction.\n", + "\n", + "The first field `ID` is the image file you did the prediction on, and the second field `annotations` is the prediction, which is further broken down into:\n", + "\n", + "- `confidences`: The percent of confidence between 0 and 1.\n", + "- `display_name`: The corresponding class name." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_the_predictions:automl,image" + }, + "outputs": [], + "source": [ + "def get_latest_predictions(gcs_out_dir):\n", + " \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n", + " folders = !gsutil ls $gcs_out_dir\n", + " latest = \"\"\n", + " for folder in folders:\n", + " subfolder = folder.split(\"/\")[-2]\n", + " if subfolder.startswith(\"prediction-\"):\n", + " if subfolder > latest:\n", + " latest = folder[:-1]\n", + " return latest\n", + "\n", + "\n", + "while True:\n", + " predictions, state = get_batch_prediction_job(batch_job_id, True)\n", + " if state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", state)\n", + " if state == aip.JobState.JOB_STATE_FAILED:\n", + " raise Exception(\"Batch Job Failed\")\n", + " else:\n", + " folder = get_latest_predictions(predictions)\n", + " ! gsutil ls $folder/prediction*.jsonl\n", + "\n", + " ! gsutil cat $folder/prediction*.jsonl\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_automl_image_classification_batch.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/automl/showcase_automl_image_classification_export_edge.ipynb b/notebooks/community/gapic/automl/showcase_automl_image_classification_export_edge.ipynb new file mode 100644 index 000000000..8bbbefa23 --- /dev/null +++ b/notebooks/community/gapic/automl/showcase_automl_image_classification_export_edge.ipynb @@ -0,0 +1,1468 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: AutoML image classification model for export to edge\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl,export_edge" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to create image classification models to export as an Edge model using Google Cloud's AutoML." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:flowers,icn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Flowers dataset](https://www.tensorflow.org/datasets/catalog/tf_flowers) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts the type of flower an image is from a class of five flowers: daisy, dandelion, rose, sunflower, or tulip." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,export_edge" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create a AutoML image classification model from a Python script using the Vertex client library, and then export the model as an Edge model in TFLite format. You can alternatively create models with AutoML using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- Export the `Edge` model from the `Model` resource to Cloud Storage.\n", + "- Download the model locally.\n", + "- Make a local prediction." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:export_edge" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for exporting the trained model. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Set constants unique to AutoML datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the `Dataset` resource service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the `Dataset` resource service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:icn" + }, + "outputs": [], + "source": [ + "# Image Dataset type\n", + "DATA_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/metadata/image_1.0.0.yaml\"\n", + "# Image Labeling type\n", + "LABEL_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/ioformat/image_classification_single_label_io_format_1.0.0.yaml\"\n", + "# Image Training task\n", + "TRAINING_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_image_classification_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML image classification model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:automl,export_edge" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:automl,export_edge" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "\n", + "def create_dataset(name, schema, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " dataset = aip.Dataset(\n", + " display_name=name, metadata_schema_uri=schema, labels=labels\n", + " )\n", + "\n", + " operation = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"flowers-\" + TIMESTAMP, DATA_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:image,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for images has some requirements for your data:\n", + "\n", + "- Images must be stored in a Cloud Storage bucket.\n", + "- Each image file must be in an image format (PNG, JPEG, BMP, ...).\n", + "- There must be an index file stored in your Cloud Storage bucket that contains the path and label for each image.\n", + "- The index file must be either CSV or JSONL." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:icn,u_dataset,csv" + }, + "source": [ + "#### CSV\n", + "\n", + "For image classification, the CSV index file has the requirements:\n", + "\n", + "- No heading.\n", + "- First column is the Cloud Storage path to the image.\n", + "- Second column is the label." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:flowers,csv,icn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = (\n", + " \"gs://cloud-samples-data/vision/automl_classification/flowers/all_data_v2.csv\"\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:csv" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Flowers dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:csv" + }, + "outputs": [], + "source": [ + "if \"IMPORT_FILES\" in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_data" + }, + "source": [ + "### Import data\n", + "\n", + "Now, import the data into your Vertex Dataset resource. Use this helper function `import_data` to import the data. The function does the following:\n", + "\n", + "- Uses the `Dataset` client.\n", + "- Calls the client method `import_data`, with the following parameters:\n", + " - `name`: The human readable name you give to the `Dataset` resource (e.g., flowers).\n", + " - `import_configs`: The import configuration.\n", + "\n", + "- `import_configs`: A Python list containing a dictionary, with the key/value entries:\n", + " - `gcs_sources`: A list of URIs to the paths of the one or more index files.\n", + " - `import_schema_uri`: The schema identifying the labeling type.\n", + "\n", + "The `import_data()` method returns a long running `operation` object. This will take a few minutes to complete. If you are in a live tutorial, this would be a good time to ask questions, or take a personal break." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_data" + }, + "outputs": [], + "source": [ + "def import_data(dataset, gcs_sources, schema):\n", + " config = [{\"gcs_source\": {\"uris\": gcs_sources}, \"import_schema_uri\": schema}]\n", + " print(\"dataset:\", dataset_id)\n", + " start_time = time.time()\n", + " try:\n", + " operation = clients[\"dataset\"].import_data(\n", + " name=dataset_id, import_configs=config\n", + " )\n", + " print(\"Long running operation:\", operation.operation.name)\n", + "\n", + " result = operation.result()\n", + " print(\"result:\", result)\n", + " print(\"time:\", int(time.time() - start_time), \"secs\")\n", + " print(\"error:\", operation.exception())\n", + " print(\"meta :\", operation.metadata)\n", + " print(\n", + " \"after: running:\",\n", + " operation.running(),\n", + " \"done:\",\n", + " operation.done(),\n", + " \"cancelled:\",\n", + " operation.cancelled(),\n", + " )\n", + "\n", + " return operation\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "import_data(dataset_id, [IMPORT_FILE], LABEL_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML image classification model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_pipeline:automl" + }, + "source": [ + "### Create a training pipeline\n", + "\n", + "You may ask, what do we use a pipeline for? You typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `pipeline_name`: A human readable name for the pipeline job.\n", + "- `model_name`: A human readable name for the model.\n", + "- `dataset`: The Vertex fully qualified dataset identifier.\n", + "- `schema`: The dataset labeling (annotation) training schema.\n", + "- `task`: A dictionary describing the requirements for the training job.\n", + "\n", + "The helper function calls the `Pipeline` client service'smethod `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "Let's look now deeper into the *minimal* requirements for constructing a `training_pipeline` specification:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The dataset labeling (annotation) training schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A human readable name for the model.\n", + "- `input_data_config`: The dataset specification.\n", + " - `dataset_id`: The Vertex dataset identifier only (non-fully qualified) -- this is the last part of the fully-qualified identifier.\n", + " - `fraction_split`: If specified, the percentages of the dataset to use for training, test and validation. Otherwise, the percentages are automatically selected by AutoML." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_pipeline:automl" + }, + "outputs": [], + "source": [ + "def create_pipeline(pipeline_name, model_name, dataset, schema, task):\n", + "\n", + " dataset_id = dataset.split(\"/\")[-1]\n", + "\n", + " input_config = {\n", + " \"dataset_id\": dataset_id,\n", + " \"fraction_split\": {\n", + " \"training_fraction\": 0.8,\n", + " \"validation_fraction\": 0.1,\n", + " \"test_fraction\": 0.1,\n", + " },\n", + " }\n", + "\n", + " training_pipeline = {\n", + " \"display_name\": pipeline_name,\n", + " \"training_task_definition\": schema,\n", + " \"training_task_inputs\": task,\n", + " \"input_data_config\": input_config,\n", + " \"model_to_upload\": {\"display_name\": model_name},\n", + " }\n", + "\n", + " try:\n", + " pipeline = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " )\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "task_requirements:automl,icn,edge" + }, + "source": [ + "### Construct the task requirements\n", + "\n", + "Next, construct the task requirements. Unlike other parameters which take a Python (JSON-like) dictionary, the `task` field takes a Google protobuf Struct, which is very similar to a Python dictionary. Use the `json_format.ParseDict` method for the conversion.\n", + "\n", + "The minimal fields we need to specify are:\n", + "\n", + "- `multi_label`: Whether True/False this is a multi-label (vs single) classification.\n", + "- `budget_milli_node_hours`: The maximum time to budget (billed) for training the model, where 1000 = 1 hour. For image classification, the budget must be a minimum of 8 hours.\n", + "- `model_type`: The type of deployed model:\n", + " - `CLOUD`: For deploying to Google Cloud.\n", + " - `MOBILE_TF_LOW_LATENCY_1`: For deploying to the edge and optimizing for latency (response time).\n", + " - `MOBILE_TF_HIGH_ACCURACY_1`: For deploying to the edge and optimizing for accuracy.\n", + " - `MOBILE_TF_VERSATILE_1`: For deploying to the edge and optimizing for a trade off between latency and accuracy.\n", + "- `disable_early_stopping`: Whether True/False to let AutoML use its judgement to stop training early or train for the entire budget.\n", + "\n", + "Finally, create the pipeline by calling the helper function `create_pipeline`, which returns an instance of a training pipeline object." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_requirements:automl,icn,edge" + }, + "outputs": [], + "source": [ + "PIPE_NAME = \"flowers_pipe-\" + TIMESTAMP\n", + "MODEL_NAME = \"flowers_model-\" + TIMESTAMP\n", + "\n", + "task = json_format.ParseDict(\n", + " {\n", + " \"multi_label\": False,\n", + " \"budget_milli_node_hours\": 8000,\n", + " \"model_type\": \"MOBILE_TF_LOW_LATENCY_1\",\n", + " \"disable_early_stopping\": False,\n", + " },\n", + " Value(),\n", + ")\n", + "\n", + "response = create_pipeline(PIPE_NAME, MODEL_NAME, dataset_id, TRAINING_SCHEMA, task)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 30 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "model_information" + }, + "source": [ + "## Model information\n", + "\n", + "Now that your model is trained, you can get some information on your model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:automl" + }, + "source": [ + "## Evaluate the Model resource\n", + "\n", + "Now find out how good the model service believes your model is. As part of training, some portion of the dataset was set aside as the test (holdout) data, which is used by the pipeline service to evaluate the model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "list_model_evaluations:automl,icn" + }, + "source": [ + "### List evaluations for all slices\n", + "\n", + "Use this helper function `list_model_evaluations`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified model identifier for the `Model` resource.\n", + "\n", + "This helper function uses the model client service's `list_model_evaluations` method, which takes the same parameter. The response object from the call is a list, where each element is an evaluation metric.\n", + "\n", + "For each evaluation (you probably only have one) we then print all the key names for each metric in the evaluation, and for a small set (`logLoss` and `auPrc`) you will print the result." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "list_model_evaluations:automl,icn" + }, + "outputs": [], + "source": [ + "def list_model_evaluations(name):\n", + " response = clients[\"model\"].list_model_evaluations(parent=name)\n", + " for evaluation in response:\n", + " print(\"model_evaluation\")\n", + " print(\" name:\", evaluation.name)\n", + " print(\" metrics_schema_uri:\", evaluation.metrics_schema_uri)\n", + " metrics = json_format.MessageToDict(evaluation._pb.metrics)\n", + " for metric in metrics.keys():\n", + " print(metric)\n", + " print(\"logloss\", metrics[\"logLoss\"])\n", + " print(\"auPrc\", metrics[\"auPrc\"])\n", + "\n", + " return evaluation.name\n", + "\n", + "\n", + "last_evaluation = list_model_evaluations(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "export_model:export_edge" + }, + "source": [ + "## Export as Edge model\n", + "\n", + "You can export an AutoML image classification model as an Edge model which you can then custom deploy to an edge device, such as a mobile phone or IoT device, or download locally. Use this helper function `export_model` to export the model to Google Cloud, which takes the following parameters:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `format`: The format to save the model format as.\n", + "- `gcs_dest`: The Cloud Storage location to store the SavedFormat model artifacts to.\n", + "\n", + "This function calls the `Model` client service's method `export_model`, with the following parameters:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `output_config`: The destination information for the exported model.\n", + " - `artifact_destination.output_uri_prefix`: The Cloud Storage location to store the SavedFormat model artifacts to.\n", + " - `export_format_id`: The format to save the model format as. For AutoML image classification:\n", + " - `tf-saved-model`: TensorFlow SavedFormat for deployment to a container.\n", + " - `tflite`: TensorFlow Lite for deployment to an edge or mobile device.\n", + " - `edgetpu-tflite`: TensorFlow Lite for TPU\n", + " - `tf-js`: TensorFlow for web client\n", + " - `coral-ml`: for Coral devices\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is exported." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "export_model:export_edge" + }, + "outputs": [], + "source": [ + "MODEL_DIR = BUCKET_NAME + \"/\" + \"flowers\"\n", + "\n", + "\n", + "def export_model(name, format, gcs_dest):\n", + " output_config = {\n", + " \"artifact_destination\": {\"output_uri_prefix\": gcs_dest},\n", + " \"export_format_id\": format,\n", + " }\n", + " response = clients[\"model\"].export_model(name=name, output_config=output_config)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result(timeout=1800)\n", + " metadata = response.operation.metadata\n", + " artifact_uri = str(metadata.value).split(\"\\\\\")[-1][4:-1]\n", + " print(\"Artifact Uri\", artifact_uri)\n", + " return artifact_uri\n", + "\n", + "\n", + "model_package = export_model(model_to_deploy_id, \"tflite\", MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "download_model_artifacts:tflite" + }, + "source": [ + "#### Download the TFLite model artifacts\n", + "\n", + "Now that you have an exported TFLite version of your model, you can test the exported model locally, but first downloading it from Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "download_model_artifacts:tflite" + }, + "outputs": [], + "source": [ + "! gsutil ls $model_package\n", + "# Download the model artifacts\n", + "! gsutil cp -r $model_package tflite\n", + "\n", + "tflite_path = \"tflite/model.tflite\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instantiate_tflite_interpreter" + }, + "source": [ + "#### Instantiate a TFLite interpreter\n", + "\n", + "The TFLite version of the model is not a TensorFlow SavedModel format. You cannot directly use methods like predict(). Instead, one uses the TFLite interpreter. You must first setup the interpreter for the TFLite model as follows:\n", + "\n", + "- Instantiate an TFLite interpreter for the TFLite model.\n", + "- Instruct the interpreter to allocate input and output tensors for the model.\n", + "- Get detail information about the models input and output tensors that will need to be known for prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instantiate_tflite_interpreter" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "interpreter = tf.lite.Interpreter(model_path=tflite_path)\n", + "interpreter.allocate_tensors()\n", + "\n", + "input_details = interpreter.get_input_details()\n", + "output_details = interpreter.get_output_details()\n", + "input_shape = input_details[0][\"shape\"]\n", + "\n", + "print(\"input tensor shape\", input_shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_item" + }, + "source": [ + "### Get test item\n", + "\n", + "You will use an arbitrary example out of the dataset as a test item. Don't be concerned that the example was likely used in training the model -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:image,224x224" + }, + "outputs": [], + "source": [ + "test_items = ! gsutil cat $IMPORT_FILE | head -n1\n", + "test_item = test_items[0].split(\",\")[0]\n", + "\n", + "with tf.io.gfile.GFile(test_item, \"rb\") as f:\n", + " content = f.read()\n", + "test_image = tf.io.decode_jpeg(content)\n", + "print(\"test image shape\", test_image.shape)\n", + "\n", + "test_image = tf.image.resize(test_image, (224, 224))\n", + "print(\"test image shape\", test_image.shape, test_image.dtype)\n", + "\n", + "test_image = tf.cast(test_image, dtype=tf.uint8).numpy()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "invoke_tflite_interpreter" + }, + "source": [ + "#### Make a prediction with TFLite model\n", + "\n", + "Finally, you do a prediction using your TFLite model, as follows:\n", + "\n", + "- Convert the test image into a batch of a single image (`np.expand_dims`)\n", + "- Set the input tensor for the interpreter to your batch of a single image (`data`).\n", + "- Invoke the interpreter.\n", + "- Retrieve the softmax probabilities for the prediction (`get_tensor`).\n", + "- Determine which label had the highest probability (`np.argmax`)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "invoke_tflite_interpreter" + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "\n", + "data = np.expand_dims(test_image, axis=0)\n", + "\n", + "interpreter.set_tensor(input_details[0][\"index\"], data)\n", + "\n", + "interpreter.invoke()\n", + "\n", + "softmax = interpreter.get_tensor(output_details[0][\"index\"])\n", + "\n", + "label = np.argmax(softmax)\n", + "\n", + "print(label)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_automl_image_classification_export_edge.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/automl/showcase_automl_image_classification_online.ipynb b/notebooks/community/gapic/automl/showcase_automl_image_classification_online.ipynb new file mode 100644 index 000000000..8e8edb9b4 --- /dev/null +++ b/notebooks/community/gapic/automl/showcase_automl_image_classification_online.ipynb @@ -0,0 +1,1587 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: AutoML image classification model for online prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to create image classification models and do online prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:flowers,icn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Flowers dataset](https://www.tensorflow.org/datasets/catalog/tf_flowers) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts the type of flower an image is from a class of five flowers: daisy, dandelion, rose, sunflower, or tulip." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,online_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML image classification model and deploy for online prediction from a Python script using the Vertex client library. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model`." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Set constants unique to AutoML datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the `Dataset` resource service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the `Dataset` resource service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:icn" + }, + "outputs": [], + "source": [ + "# Image Dataset type\n", + "DATA_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/metadata/image_1.0.0.yaml\"\n", + "# Image Labeling type\n", + "LABEL_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/ioformat/image_classification_single_label_io_format_1.0.0.yaml\"\n", + "# Image Training task\n", + "TRAINING_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_image_classification_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML image classification model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:automl,online_prediction" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training.\n", + "- Endpoint Service for deployment.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:automl,online_prediction" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "\n", + "def create_dataset(name, schema, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " dataset = aip.Dataset(\n", + " display_name=name, metadata_schema_uri=schema, labels=labels\n", + " )\n", + "\n", + " operation = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"flowers-\" + TIMESTAMP, DATA_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:image,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for images has some requirements for your data:\n", + "\n", + "- Images must be stored in a Cloud Storage bucket.\n", + "- Each image file must be in an image format (PNG, JPEG, BMP, ...).\n", + "- There must be an index file stored in your Cloud Storage bucket that contains the path and label for each image.\n", + "- The index file must be either CSV or JSONL." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:icn,u_dataset,csv" + }, + "source": [ + "#### CSV\n", + "\n", + "For image classification, the CSV index file has the requirements:\n", + "\n", + "- No heading.\n", + "- First column is the Cloud Storage path to the image.\n", + "- Second column is the label." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:flowers,csv,icn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = (\n", + " \"gs://cloud-samples-data/vision/automl_classification/flowers/all_data_v2.csv\"\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:csv" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Flowers dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:csv" + }, + "outputs": [], + "source": [ + "if \"IMPORT_FILES\" in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_data" + }, + "source": [ + "### Import data\n", + "\n", + "Now, import the data into your Vertex Dataset resource. Use this helper function `import_data` to import the data. The function does the following:\n", + "\n", + "- Uses the `Dataset` client.\n", + "- Calls the client method `import_data`, with the following parameters:\n", + " - `name`: The human readable name you give to the `Dataset` resource (e.g., flowers).\n", + " - `import_configs`: The import configuration.\n", + "\n", + "- `import_configs`: A Python list containing a dictionary, with the key/value entries:\n", + " - `gcs_sources`: A list of URIs to the paths of the one or more index files.\n", + " - `import_schema_uri`: The schema identifying the labeling type.\n", + "\n", + "The `import_data()` method returns a long running `operation` object. This will take a few minutes to complete. If you are in a live tutorial, this would be a good time to ask questions, or take a personal break." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_data" + }, + "outputs": [], + "source": [ + "def import_data(dataset, gcs_sources, schema):\n", + " config = [{\"gcs_source\": {\"uris\": gcs_sources}, \"import_schema_uri\": schema}]\n", + " print(\"dataset:\", dataset_id)\n", + " start_time = time.time()\n", + " try:\n", + " operation = clients[\"dataset\"].import_data(\n", + " name=dataset_id, import_configs=config\n", + " )\n", + " print(\"Long running operation:\", operation.operation.name)\n", + "\n", + " result = operation.result()\n", + " print(\"result:\", result)\n", + " print(\"time:\", int(time.time() - start_time), \"secs\")\n", + " print(\"error:\", operation.exception())\n", + " print(\"meta :\", operation.metadata)\n", + " print(\n", + " \"after: running:\",\n", + " operation.running(),\n", + " \"done:\",\n", + " operation.done(),\n", + " \"cancelled:\",\n", + " operation.cancelled(),\n", + " )\n", + "\n", + " return operation\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "import_data(dataset_id, [IMPORT_FILE], LABEL_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML image classification model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_pipeline:automl" + }, + "source": [ + "### Create a training pipeline\n", + "\n", + "You may ask, what do we use a pipeline for? You typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `pipeline_name`: A human readable name for the pipeline job.\n", + "- `model_name`: A human readable name for the model.\n", + "- `dataset`: The Vertex fully qualified dataset identifier.\n", + "- `schema`: The dataset labeling (annotation) training schema.\n", + "- `task`: A dictionary describing the requirements for the training job.\n", + "\n", + "The helper function calls the `Pipeline` client service'smethod `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "Let's look now deeper into the *minimal* requirements for constructing a `training_pipeline` specification:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The dataset labeling (annotation) training schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A human readable name for the model.\n", + "- `input_data_config`: The dataset specification.\n", + " - `dataset_id`: The Vertex dataset identifier only (non-fully qualified) -- this is the last part of the fully-qualified identifier.\n", + " - `fraction_split`: If specified, the percentages of the dataset to use for training, test and validation. Otherwise, the percentages are automatically selected by AutoML." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_pipeline:automl" + }, + "outputs": [], + "source": [ + "def create_pipeline(pipeline_name, model_name, dataset, schema, task):\n", + "\n", + " dataset_id = dataset.split(\"/\")[-1]\n", + "\n", + " input_config = {\n", + " \"dataset_id\": dataset_id,\n", + " \"fraction_split\": {\n", + " \"training_fraction\": 0.8,\n", + " \"validation_fraction\": 0.1,\n", + " \"test_fraction\": 0.1,\n", + " },\n", + " }\n", + "\n", + " training_pipeline = {\n", + " \"display_name\": pipeline_name,\n", + " \"training_task_definition\": schema,\n", + " \"training_task_inputs\": task,\n", + " \"input_data_config\": input_config,\n", + " \"model_to_upload\": {\"display_name\": model_name},\n", + " }\n", + "\n", + " try:\n", + " pipeline = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " )\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "task_requirements:automl,icn" + }, + "source": [ + "### Construct the task requirements\n", + "\n", + "Next, construct the task requirements. Unlike other parameters which take a Python (JSON-like) dictionary, the `task` field takes a Google protobuf Struct, which is very similar to a Python dictionary. Use the `json_format.ParseDict` method for the conversion.\n", + "\n", + "The minimal fields we need to specify are:\n", + "\n", + "- `multi_label`: Whether True/False this is a multi-label (vs single) classification.\n", + "- `budget_milli_node_hours`: The maximum time to budget (billed) for training the model, where 1000 = 1 hour. For image classification, the budget must be a minimum of 8 hours.\n", + "- `model_type`: The type of deployed model:\n", + " - `CLOUD`: For deploying to Google Cloud.\n", + " - `MOBILE_TF_LOW_LATENCY_1`: For deploying to the edge and optimizing for latency (response time).\n", + " - `MOBILE_TF_HIGH_ACCURACY_1`: For deploying to the edge and optimizing for accuracy.\n", + " - `MOBILE_TF_VERSATILE_1`: For deploying to the edge and optimizing for a trade off between latency and accuracy.\n", + "- `disable_early_stopping`: Whether True/False to let AutoML use its judgement to stop training early or train for the entire budget.\n", + "\n", + "Finally, create the pipeline by calling the helper function `create_pipeline`, which returns an instance of a training pipeline object." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_requirements:automl,icn" + }, + "outputs": [], + "source": [ + "PIPE_NAME = \"flowers_pipe-\" + TIMESTAMP\n", + "MODEL_NAME = \"flowers_model-\" + TIMESTAMP\n", + "\n", + "task = json_format.ParseDict(\n", + " {\n", + " \"multi_label\": False,\n", + " \"budget_milli_node_hours\": 8000,\n", + " \"model_type\": \"CLOUD\",\n", + " \"disable_early_stopping\": False,\n", + " },\n", + " Value(),\n", + ")\n", + "\n", + "response = create_pipeline(PIPE_NAME, MODEL_NAME, dataset_id, TRAINING_SCHEMA, task)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "model_information" + }, + "source": [ + "## Model information\n", + "\n", + "Now that your model is trained, you can get some information on your model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:automl" + }, + "source": [ + "## Evaluate the Model resource\n", + "\n", + "Now find out how good the model service believes your model is. As part of training, some portion of the dataset was set aside as the test (holdout) data, which is used by the pipeline service to evaluate the model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "list_model_evaluations:automl,icn" + }, + "source": [ + "### List evaluations for all slices\n", + "\n", + "Use this helper function `list_model_evaluations`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified model identifier for the `Model` resource.\n", + "\n", + "This helper function uses the model client service's `list_model_evaluations` method, which takes the same parameter. The response object from the call is a list, where each element is an evaluation metric.\n", + "\n", + "For each evaluation (you probably only have one) we then print all the key names for each metric in the evaluation, and for a small set (`logLoss` and `auPrc`) you will print the result." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "list_model_evaluations:automl,icn" + }, + "outputs": [], + "source": [ + "def list_model_evaluations(name):\n", + " response = clients[\"model\"].list_model_evaluations(parent=name)\n", + " for evaluation in response:\n", + " print(\"model_evaluation\")\n", + " print(\" name:\", evaluation.name)\n", + " print(\" metrics_schema_uri:\", evaluation.metrics_schema_uri)\n", + " metrics = json_format.MessageToDict(evaluation._pb.metrics)\n", + " for metric in metrics.keys():\n", + " print(metric)\n", + " print(\"logloss\", metrics[\"logLoss\"])\n", + " print(\"auPrc\", metrics[\"auPrc\"])\n", + "\n", + " return evaluation.name\n", + "\n", + "\n", + "last_evaluation = list_model_evaluations(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:automl" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created with AutoML. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"flowers_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:automatic" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `automatic_resources`: This refers to how many redundant compute instances (replicas). For this example, we set it to one (no replication).\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:automatic" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"flowers_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"automatic_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " },\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_item" + }, + "source": [ + "### Get test item\n", + "\n", + "You will use an arbitrary example out of the dataset as a test item. Don't be concerned that the example was likely used in training the model -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:automl,icn,csv" + }, + "outputs": [], + "source": [ + "test_item = !gsutil cat $IMPORT_FILE | head -n1\n", + "if len(str(test_item[0]).split(\",\")) == 3:\n", + " _, test_item, test_label = str(test_item[0]).split(\",\")\n", + "else:\n", + " test_item, test_label = str(test_item[0]).split(\",\")\n", + "\n", + "print(test_item, test_label)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "predict_item:automl,icn" + }, + "source": [ + "### Make a prediction\n", + "\n", + "Now you have a test item. Use this helper function `predict_item`, which takes the following parameters:\n", + "\n", + "- `filename`: The Cloud Storage path to the test item.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `parameters_dict`: Additional filtering parameters for serving prediction results.\n", + "\n", + "This function calls the prediction client service's `predict` method with the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `instances`: A list of instances (encoded images) to predict.\n", + "- `parameters`: Additional filtering parameters for serving prediction results.\n", + " - `confidence_threshold`: The threshold for returning predictions. Must be between 0 and 1.\n", + " - `max_predictions`: The maximum number of predictions to return, sorted by confidence.\n", + "\n", + "How does confidence_threshold affect the model accuracy? The threshold won't change the accuracy. What it changes is *recall* and *precision*.\n", + "\n", + " - Precision: The higher the precision the more likely what is predicted is the correct prediction, but return fewer predictions. Increasing the confidence threshold increases precision.\n", + " - Recall: The higher the recall the more likely a correct prediction is returned in the result, but return more prediction with incorrect prediction. Decreasing the confidence threshold increases recall.\n", + "\n", + "In this example, you will predict for precision. You set the confidence threshold to 0.5 and the maximum number of predictions for a classification to two. Since, all the confidence values across the classes must add up to one, there are only two possible outcomes:\n", + "\n", + " 1. There is a tie, both 0.5, and returns two predictions.\n", + " 2. One value is above 0.5 and the rest are below 0.5, and returns one prediction.\n", + "\n", + "#### Request\n", + "\n", + "Since in this example your test item is in a Cloud Storage bucket, you open and read the contents of the image using `tf.io.gfile.Gfile()`. To pass the test data to the prediction service, you encode the bytes into base64 -- which makes the content safe from modification while transmitting binary data over the network.\n", + "\n", + "The format of each instance is:\n", + "\n", + " { 'content': { 'b64': [base64_encoded_bytes] } }\n", + "\n", + "Since the `predict()` method can take multiple items (instances), send your single test item as a list of one test item. As a final step, you package the instances list into Google's protobuf format -- which is what you pass to the `predict()` method.\n", + "\n", + "#### Response\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding image in the request. You will see in the output for each prediction (just one in this case):\n", + "\n", + "- `ids`: The instance ID of each data item.\n", + "- `confidences`: The percent of confidence between 0 and 1 in the prediction for each class.\n", + "- `displayNames`: The corresponding class names." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "predict_item:automl,icn" + }, + "outputs": [], + "source": [ + "import base64\n", + "\n", + "import tensorflow as tf\n", + "\n", + "\n", + "def predict_item(filename, endpoint, parameters_dict):\n", + "\n", + " parameters = json_format.ParseDict(parameters_dict, Value())\n", + " with tf.io.gfile.GFile(filename, \"rb\") as f:\n", + " content = f.read()\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [{\"content\": base64.b64encode(content).decode(\"utf-8\")}]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", dict(prediction))\n", + "\n", + "\n", + "predict_item(test_item, endpoint_id, {\"confidenceThreshold\": 0.5, \"maxPredictions\": 2})" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_automl_image_classification_online.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/automl/showcase_automl_image_classification_online_proxy.ipynb b/notebooks/community/gapic/automl/showcase_automl_image_classification_online_proxy.ipynb new file mode 100644 index 000000000..45dd0d32d --- /dev/null +++ b/notebooks/community/gapic/automl/showcase_automl_image_classification_online_proxy.ipynb @@ -0,0 +1,1766 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: AutoML image classification model for online prediction using Cloud Function\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to create image classification models and do online prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:flowers,icn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Flowers dataset](https://www.tensorflow.org/datasets/catalog/tf_flowers) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts the type of flower an image is from a class of five flowers: daisy, dandelion, rose, sunflower, or tulip." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,online_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML image classification model and deploy for online prediction from a Python script using the Vertex client library. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model`." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Set constants unique to AutoML datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the `Dataset` resource service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the `Dataset` resource service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:icn" + }, + "outputs": [], + "source": [ + "# Image Dataset type\n", + "DATA_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/metadata/image_1.0.0.yaml\"\n", + "# Image Labeling type\n", + "LABEL_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/ioformat/image_classification_single_label_io_format_1.0.0.yaml\"\n", + "# Image Training task\n", + "TRAINING_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_image_classification_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML image classification model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:automl,online_prediction" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training.\n", + "- Endpoint Service for deployment.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:automl,online_prediction" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "\n", + "def create_dataset(name, schema, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " dataset = aip.Dataset(\n", + " display_name=name, metadata_schema_uri=schema, labels=labels\n", + " )\n", + "\n", + " operation = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"flowers-\" + TIMESTAMP, DATA_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:image,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for images has some requirements for your data:\n", + "\n", + "- Images must be stored in a Cloud Storage bucket.\n", + "- Each image file must be in an image format (PNG, JPEG, BMP, ...).\n", + "- There must be an index file stored in your Cloud Storage bucket that contains the path and label for each image.\n", + "- The index file must be either CSV or JSONL." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:icn,u_dataset,csv" + }, + "source": [ + "#### CSV\n", + "\n", + "For image classification, the CSV index file has the requirements:\n", + "\n", + "- No heading.\n", + "- First column is the Cloud Storage path to the image.\n", + "- Second column is the label." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:flowers,csv,icn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = (\n", + " \"gs://cloud-samples-data/vision/automl_classification/flowers/all_data_v2.csv\"\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:csv" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Flowers dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:csv" + }, + "outputs": [], + "source": [ + "if \"IMPORT_FILES\" in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_data" + }, + "source": [ + "### Import data\n", + "\n", + "Now, import the data into your Vertex Dataset resource. Use this helper function `import_data` to import the data. The function does the following:\n", + "\n", + "- Uses the `Dataset` client.\n", + "- Calls the client method `import_data`, with the following parameters:\n", + " - `name`: The human readable name you give to the `Dataset` resource (e.g., flowers).\n", + " - `import_configs`: The import configuration.\n", + "\n", + "- `import_configs`: A Python list containing a dictionary, with the key/value entries:\n", + " - `gcs_sources`: A list of URIs to the paths of the one or more index files.\n", + " - `import_schema_uri`: The schema identifying the labeling type.\n", + "\n", + "The `import_data()` method returns a long running `operation` object. This will take a few minutes to complete. If you are in a live tutorial, this would be a good time to ask questions, or take a personal break." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_data" + }, + "outputs": [], + "source": [ + "def import_data(dataset, gcs_sources, schema):\n", + " config = [{\"gcs_source\": {\"uris\": gcs_sources}, \"import_schema_uri\": schema}]\n", + " print(\"dataset:\", dataset_id)\n", + " start_time = time.time()\n", + " try:\n", + " operation = clients[\"dataset\"].import_data(\n", + " name=dataset_id, import_configs=config\n", + " )\n", + " print(\"Long running operation:\", operation.operation.name)\n", + "\n", + " result = operation.result()\n", + " print(\"result:\", result)\n", + " print(\"time:\", int(time.time() - start_time), \"secs\")\n", + " print(\"error:\", operation.exception())\n", + " print(\"meta :\", operation.metadata)\n", + " print(\n", + " \"after: running:\",\n", + " operation.running(),\n", + " \"done:\",\n", + " operation.done(),\n", + " \"cancelled:\",\n", + " operation.cancelled(),\n", + " )\n", + "\n", + " return operation\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "import_data(dataset_id, [IMPORT_FILE], LABEL_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML image classification model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_pipeline:automl" + }, + "source": [ + "### Create a training pipeline\n", + "\n", + "You may ask, what do we use a pipeline for? You typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `pipeline_name`: A human readable name for the pipeline job.\n", + "- `model_name`: A human readable name for the model.\n", + "- `dataset`: The Vertex fully qualified dataset identifier.\n", + "- `schema`: The dataset labeling (annotation) training schema.\n", + "- `task`: A dictionary describing the requirements for the training job.\n", + "\n", + "The helper function calls the `Pipeline` client service'smethod `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "Let's look now deeper into the *minimal* requirements for constructing a `training_pipeline` specification:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The dataset labeling (annotation) training schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A human readable name for the model.\n", + "- `input_data_config`: The dataset specification.\n", + " - `dataset_id`: The Vertex dataset identifier only (non-fully qualified) -- this is the last part of the fully-qualified identifier.\n", + " - `fraction_split`: If specified, the percentages of the dataset to use for training, test and validation. Otherwise, the percentages are automatically selected by AutoML." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_pipeline:automl" + }, + "outputs": [], + "source": [ + "def create_pipeline(pipeline_name, model_name, dataset, schema, task):\n", + "\n", + " dataset_id = dataset.split(\"/\")[-1]\n", + "\n", + " input_config = {\n", + " \"dataset_id\": dataset_id,\n", + " \"fraction_split\": {\n", + " \"training_fraction\": 0.8,\n", + " \"validation_fraction\": 0.1,\n", + " \"test_fraction\": 0.1,\n", + " },\n", + " }\n", + "\n", + " training_pipeline = {\n", + " \"display_name\": pipeline_name,\n", + " \"training_task_definition\": schema,\n", + " \"training_task_inputs\": task,\n", + " \"input_data_config\": input_config,\n", + " \"model_to_upload\": {\"display_name\": model_name},\n", + " }\n", + "\n", + " try:\n", + " pipeline = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " )\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "task_requirements:automl,icn" + }, + "source": [ + "### Construct the task requirements\n", + "\n", + "Next, construct the task requirements. Unlike other parameters which take a Python (JSON-like) dictionary, the `task` field takes a Google protobuf Struct, which is very similar to a Python dictionary. Use the `json_format.ParseDict` method for the conversion.\n", + "\n", + "The minimal fields we need to specify are:\n", + "\n", + "- `multi_label`: Whether True/False this is a multi-label (vs single) classification.\n", + "- `budget_milli_node_hours`: The maximum time to budget (billed) for training the model, where 1000 = 1 hour. For image classification, the budget must be a minimum of 8 hours.\n", + "- `model_type`: The type of deployed model:\n", + " - `CLOUD`: For deploying to Google Cloud.\n", + " - `MOBILE_TF_LOW_LATENCY_1`: For deploying to the edge and optimizing for latency (response time).\n", + " - `MOBILE_TF_HIGH_ACCURACY_1`: For deploying to the edge and optimizing for accuracy.\n", + " - `MOBILE_TF_VERSATILE_1`: For deploying to the edge and optimizing for a trade off between latency and accuracy.\n", + "- `disable_early_stopping`: Whether True/False to let AutoML use its judgement to stop training early or train for the entire budget.\n", + "\n", + "Finally, create the pipeline by calling the helper function `create_pipeline`, which returns an instance of a training pipeline object." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_requirements:automl,icn" + }, + "outputs": [], + "source": [ + "PIPE_NAME = \"flowers_pipe-\" + TIMESTAMP\n", + "MODEL_NAME = \"flowers_model-\" + TIMESTAMP\n", + "\n", + "task = json_format.ParseDict(\n", + " {\n", + " \"multi_label\": False,\n", + " \"budget_milli_node_hours\": 8000,\n", + " \"model_type\": \"CLOUD\",\n", + " \"disable_early_stopping\": False,\n", + " },\n", + " Value(),\n", + ")\n", + "\n", + "response = create_pipeline(PIPE_NAME, MODEL_NAME, dataset_id, TRAINING_SCHEMA, task)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "model_information" + }, + "source": [ + "## Model information\n", + "\n", + "Now that your model is trained, you can get some information on your model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:automl" + }, + "source": [ + "## Evaluate the Model resource\n", + "\n", + "Now find out how good the model service believes your model is. As part of training, some portion of the dataset was set aside as the test (holdout) data, which is used by the pipeline service to evaluate the model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "list_model_evaluations:automl,icn" + }, + "source": [ + "### List evaluations for all slices\n", + "\n", + "Use this helper function `list_model_evaluations`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified model identifier for the `Model` resource.\n", + "\n", + "This helper function uses the model client service's `list_model_evaluations` method, which takes the same parameter. The response object from the call is a list, where each element is an evaluation metric.\n", + "\n", + "For each evaluation (you probably only have one) we then print all the key names for each metric in the evaluation, and for a small set (`logLoss` and `auPrc`) you will print the result." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "list_model_evaluations:automl,icn" + }, + "outputs": [], + "source": [ + "def list_model_evaluations(name):\n", + " response = clients[\"model\"].list_model_evaluations(parent=name)\n", + " for evaluation in response:\n", + " print(\"model_evaluation\")\n", + " print(\" name:\", evaluation.name)\n", + " print(\" metrics_schema_uri:\", evaluation.metrics_schema_uri)\n", + " metrics = json_format.MessageToDict(evaluation._pb.metrics)\n", + " for metric in metrics.keys():\n", + " print(metric)\n", + " print(\"logloss\", metrics[\"logLoss\"])\n", + " print(\"auPrc\", metrics[\"auPrc\"])\n", + "\n", + " return evaluation.name\n", + "\n", + "\n", + "last_evaluation = list_model_evaluations(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:automl" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created with AutoML. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"flowers_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:automatic" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `automatic_resources`: This refers to how many redundant compute instances (replicas). For this example, we set it to one (no replication).\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:automatic" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"flowers_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"automatic_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " },\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_item" + }, + "source": [ + "### Get test item\n", + "\n", + "You will use an arbitrary example out of the dataset as a test item. Don't be concerned that the example was likely used in training the model -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:automl,icn,csv" + }, + "outputs": [], + "source": [ + "test_item = !gsutil cat $IMPORT_FILE | head -n1\n", + "if len(str(test_item[0]).split(\",\")) == 3:\n", + " _, test_item, test_label = str(test_item[0]).split(\",\")\n", + "else:\n", + " test_item, test_label = str(test_item[0]).split(\",\")\n", + "\n", + "print(test_item, test_label)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_cloud_function_for_endpoint" + }, + "source": [ + "## Predict through Cloud Function proxy\n", + "\n", + "Endpoints in Vertex are pinned to the project they are deployed in. That means, in general, only process operating within the same project can access the endpoint for a prediction.\n", + "\n", + "There are several approaches for a web-based application to obtain predictions from an endpoint:\n", + "\n", + "- The backend application server for the web-based application is deployed in the same project.\n", + "\n", + "- The backend application server is deployed in a different project which has access rights granted to the project where the endpoint is deployed.\n", + "\n", + "- Use a proxy which as access rights to the project where the endpoint is deployed.\n", + "\n", + "In this example, the proxy method is demonstrated using Cloud Functions.\n", + "\n", + "### Create a cloud function\n", + "\n", + "You start by creating a cloud function, which will consists of:\n", + "\n", + "- A folder (e.g., function)\n", + " - main.py: The implemented function as a Python script.\n", + " - requirements.txt: The environment requirements for executing the Python script." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_cloud_function_for_endpoint" + }, + "outputs": [], + "source": [ + "! rm -rf function\n", + "! mkdir function" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_cloud_function_for_endpoint:flowers" + }, + "outputs": [], + "source": [ + "%%writefile function/main.py\n", + "import logging\n", + "from operator import itemgetter\n", + "import os\n", + "\n", + "from flask import jsonify\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.struct_pb2 import Value\n", + "import requests\n", + "import tensorflow as tf\n", + "\n", + "IMG_WIDTH = 128\n", + "COLUMNS = ['dandelion', 'daisy', 'tulips', 'sunflowers', 'roses']\n", + "\n", + "aip_client = aip.PredictionServiceClient(client_options={\n", + " 'api_endpoint': 'us-central1-prediction-aiplatform.googleapis.com'\n", + "})\n", + "aip_endpoint_name = f'projects/{os.environ[\"GCP_PROJECT\"]}/locations/us-central1/endpoints/{os.environ[\"ENDPOINT_ID\"]}'\n", + "\n", + "\n", + "def get_prediction(instance):\n", + " logging.info('Sending prediction request to Vertex ...')\n", + " try:\n", + " pb_instance = json_format.ParseDict(instance, Value())\n", + " response = aip_client.predict(endpoint=aip_endpoint_name,\n", + " instances=[pb_instance])\n", + " return list(response.predictions[0])\n", + " except Exception as err:\n", + " logging.error(f'Prediction request failed: {type(err)}: {err}')\n", + " return None\n", + "\n", + "\n", + "def preprocess_image(image_url):\n", + " logging.info(f'Fetching image from URL: {image_url}')\n", + " try:\n", + " image_response = requests.get(image_url)\n", + " image_response.raise_for_status()\n", + " assert image_response.headers.get('Content-Type') == 'image/jpeg'\n", + " except (ConnectionError, requests.exceptions.RequestException,\n", + " AssertionError):\n", + " logging.error(f'Error fetching image from URL: {image_url}')\n", + " return None\n", + "\n", + " logging.info('Decoding and preprocessing image ...')\n", + " image = tf.io.decode_jpeg(image_response.content, channels=3)\n", + " image = tf.image.resize_with_pad(image, IMG_WIDTH, IMG_WIDTH)\n", + " image = image / 255.\n", + " return image.numpy().tolist() # Make it JSON-serializable\n", + "\n", + "def classify_flower(request):\n", + " # Set CORS headers for the preflight request\n", + " if request.method == 'OPTIONS':\n", + " # Allows POST requests from any origin with the Content-Type\n", + " # header and caches preflight response for an 3600s\n", + " headers = {\n", + " 'Access-Control-Allow-Origin': '*',\n", + " 'Access-Control-Allow-Methods': 'POST',\n", + " 'Access-Control-Allow-Headers': 'Content-Type',\n", + " 'Access-Control-Max-Age': '3600'\n", + " }\n", + " return ('', 204, headers)\n", + "\n", + " # Disallow non-POSTs\n", + " if request.method != 'POST':\n", + " return ('Not found', 404)\n", + "\n", + " # Set CORS headers for the main request\n", + " headers = {'Access-Control-Allow-Origin': '*'}\n", + "\n", + " request_json = request.get_json(silent=True)\n", + " if not request_json or not 'image_url' in request_json:\n", + " return ('Invalid request', 400, headers)\n", + "\n", + " instance = preprocess_image(request_json['image_url'])\n", + " if not instance:\n", + " return ('Invalid request', 400, headers)\n", + "\n", + " raw_prediction = get_prediction(instance)\n", + " if not raw_prediction:\n", + " return ('Error getting prediction', 500, headers)\n", + "\n", + " probabilities = zip(COLUMNS, raw_prediction)\n", + " sorted_probabilities = sorted(probabilities,\n", + " key=itemgetter(1),\n", + " reverse=True)\n", + " return (jsonify(sorted_probabilities), 200, headers)\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_cloud_function_requirements" + }, + "outputs": [], + "source": [ + "%%writefile function/requirements.txt\n", + "Flask==1.0.2\n", + "requests==2.21.0\n", + "tensorflow-cpu~=2.1.0\n", + "google-cloud-aiplatform" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_cloud_function:flowers" + }, + "source": [ + "### Deploy your Cloud Functionl\n", + "\n", + "**TODO**\n", + "These APIs need to be enabled\n", + "Enable Cloud Function API and Cloud Build API\n", + "https://console.developers.google.com/apis/api/cloudbuild.googleapis.com/overview?project=" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_cloud_function:flowers" + }, + "outputs": [], + "source": [ + "! gcloud functions deploy classify_flower\n", + " --region $REGION \n", + " --source=function \n", + " --runtime=python37 \n", + " --memory=2048MB \n", + " --trigger-http \n", + " --allow-unauthenticated \n", + " --set-env-vars ENDPOINT_ID=${endpoint_id}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "construct_cloud_function_url" + }, + "source": [ + "#### Construct the URL to your deployed Cloud Function\n", + "\n", + "Next, you will construct the Url for your cloud function. You will use this Url to route predictions through your cloud function, acting as a proxy to your deployed model." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "construct_cloud_function_url:flowers" + }, + "outputs": [], + "source": [ + "CLOUD_FUNCTION_URL = \"https://{}-{}.cloudfunctions.net/classify_flower\".format(\n", + " REGION, PROJECT_ID\n", + ")\n", + "print(CLOUD_FUNCTION_URL)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "construct_prediction_request_json:image" + }, + "source": [ + "#### Construct prediction request file\n", + "\n", + "**TODO**" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "construct_prediction_request_json:image" + }, + "outputs": [], + "source": [ + "image = test_item.replace(\"gs://\", \"https://storage.googleapis.com/\")\n", + "print(image)\n", + "\n", + "import json\n", + "\n", + "with open(\"request.json\", \"w\") as f:\n", + " json.dump({\"image_url\": image}, f)\n", + "\n", + "! cat request.json" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "prediction_request_json" + }, + "source": [ + "#### Make the prediction request\n", + "\n", + "**TODO**" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "prediction_request_json" + }, + "outputs": [], + "source": [ + "! curl -X POST\n", + "-H \"Authorization: Bearer \"$(gcloud auth application-default print-access-token) \n", + "-H \"Content-Type: application/json; charset=utf-8\" \n", + "-d @request.json \n", + "$CLOUD_FUNCTION_URL" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_automl_image_classification_online_proxy.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/automl/showcase_automl_image_object_detection_batch.ipynb b/notebooks/community/gapic/automl/showcase_automl_image_object_detection_batch.ipynb new file mode 100644 index 000000000..a981f3361 --- /dev/null +++ b/notebooks/community/gapic/automl/showcase_automl_image_object_detection_batch.ipynb @@ -0,0 +1,1839 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: AutoML image object detection model for batch prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to create image object detection models and do batch prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:salads,iod" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the Salads category of the [OpenImages dataset](https://www.tensorflow.org/datasets/catalog/open_images_v4) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). This dataset does not require any feature engineering. The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts the bounding box locations and corresponding type of salad items in an image from a class of five items: salad, seafood, tomato, baked goods, or cheese." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,batch_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML image object detection model from a Python script, and then do a batch prediction using the Vertex client library. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Make a batch prediction.\n", + "\n", + "There is one key difference between using batch prediction and using online prediction:\n", + "\n", + "* Prediction Service: Does an on-demand prediction for the entire set of instances (i.e., one or more data items) and returns the results in real-time.\n", + "\n", + "* Batch Prediction Service: Does a queued (batch) prediction for the entire set of instances in the background and stores the results in a Cloud Storage bucket when ready." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Set constants unique to AutoML datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the `Dataset` resource service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the `Dataset` resource service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:iod" + }, + "outputs": [], + "source": [ + "# Image Dataset type\n", + "DATA_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/metadata/image_1.0.0.yaml\"\n", + "# Image Labeling type\n", + "LABEL_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/ioformat/image_bounding_box_io_format_1.0.0.yaml\"\n", + "# Image Training task\n", + "TRAINING_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_image_object_detection_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:prediction" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for prediction.\n", + "\n", + "Set the variable `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:automl" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "For AutoML batch prediction, the container image for the serving binary is pre-determined by the Vertex prediction service. More specifically, the service will pick the appropriate container for the model depending on the hardware accelerator you selected." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for prediction.\n", + "\n", + "- Set the variable `DEPLOY_COMPUTE` to configure the compute resources for the VM you will use for prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML image object detection model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:automl,batch_prediction" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training.\n", + "- Job Service for batch prediction and custom training." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:automl,batch_prediction" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"job\"] = create_job_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "\n", + "def create_dataset(name, schema, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " dataset = aip.Dataset(\n", + " display_name=name, metadata_schema_uri=schema, labels=labels\n", + " )\n", + "\n", + " operation = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"salads-\" + TIMESTAMP, DATA_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:image,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for images has some requirements for your data:\n", + "\n", + "- Images must be stored in a Cloud Storage bucket.\n", + "- Each image file must be in an image format (PNG, JPEG, BMP, ...).\n", + "- There must be an index file stored in your Cloud Storage bucket that contains the path and label for each image.\n", + "- The index file must be either CSV or JSONL." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:iod,u_dataset,csv" + }, + "source": [ + "#### CSV\n", + "\n", + "For image object detection, the CSV index file has the requirements:\n", + "\n", + "- No heading.\n", + "- First column is the Cloud Storage path to the image.\n", + "- Second column is the label.\n", + "- Third/Fourth columns are the upper left corner of bounding box. Coordinates are normalized, between 0 and 1.\n", + "- Fifth/Sixth/Seventh columns are not used and should be 0.\n", + "- Eighth/Ninth columns are the lower right corner of the bounding box." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:salads,csv,iod" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://cloud-samples-data/vision/salads.csv\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:csv" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Salads dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:csv" + }, + "outputs": [], + "source": [ + "if \"IMPORT_FILES\" in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_data" + }, + "source": [ + "### Import data\n", + "\n", + "Now, import the data into your Vertex Dataset resource. Use this helper function `import_data` to import the data. The function does the following:\n", + "\n", + "- Uses the `Dataset` client.\n", + "- Calls the client method `import_data`, with the following parameters:\n", + " - `name`: The human readable name you give to the `Dataset` resource (e.g., salads).\n", + " - `import_configs`: The import configuration.\n", + "\n", + "- `import_configs`: A Python list containing a dictionary, with the key/value entries:\n", + " - `gcs_sources`: A list of URIs to the paths of the one or more index files.\n", + " - `import_schema_uri`: The schema identifying the labeling type.\n", + "\n", + "The `import_data()` method returns a long running `operation` object. This will take a few minutes to complete. If you are in a live tutorial, this would be a good time to ask questions, or take a personal break." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_data" + }, + "outputs": [], + "source": [ + "def import_data(dataset, gcs_sources, schema):\n", + " config = [{\"gcs_source\": {\"uris\": gcs_sources}, \"import_schema_uri\": schema}]\n", + " print(\"dataset:\", dataset_id)\n", + " start_time = time.time()\n", + " try:\n", + " operation = clients[\"dataset\"].import_data(\n", + " name=dataset_id, import_configs=config\n", + " )\n", + " print(\"Long running operation:\", operation.operation.name)\n", + "\n", + " result = operation.result()\n", + " print(\"result:\", result)\n", + " print(\"time:\", int(time.time() - start_time), \"secs\")\n", + " print(\"error:\", operation.exception())\n", + " print(\"meta :\", operation.metadata)\n", + " print(\n", + " \"after: running:\",\n", + " operation.running(),\n", + " \"done:\",\n", + " operation.done(),\n", + " \"cancelled:\",\n", + " operation.cancelled(),\n", + " )\n", + "\n", + " return operation\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "import_data(dataset_id, [IMPORT_FILE], LABEL_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML image object detection model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_pipeline:automl" + }, + "source": [ + "### Create a training pipeline\n", + "\n", + "You may ask, what do we use a pipeline for? You typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `pipeline_name`: A human readable name for the pipeline job.\n", + "- `model_name`: A human readable name for the model.\n", + "- `dataset`: The Vertex fully qualified dataset identifier.\n", + "- `schema`: The dataset labeling (annotation) training schema.\n", + "- `task`: A dictionary describing the requirements for the training job.\n", + "\n", + "The helper function calls the `Pipeline` client service'smethod `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "Let's look now deeper into the *minimal* requirements for constructing a `training_pipeline` specification:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The dataset labeling (annotation) training schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A human readable name for the model.\n", + "- `input_data_config`: The dataset specification.\n", + " - `dataset_id`: The Vertex dataset identifier only (non-fully qualified) -- this is the last part of the fully-qualified identifier.\n", + " - `fraction_split`: If specified, the percentages of the dataset to use for training, test and validation. Otherwise, the percentages are automatically selected by AutoML." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_pipeline:automl" + }, + "outputs": [], + "source": [ + "def create_pipeline(pipeline_name, model_name, dataset, schema, task):\n", + "\n", + " dataset_id = dataset.split(\"/\")[-1]\n", + "\n", + " input_config = {\n", + " \"dataset_id\": dataset_id,\n", + " \"fraction_split\": {\n", + " \"training_fraction\": 0.8,\n", + " \"validation_fraction\": 0.1,\n", + " \"test_fraction\": 0.1,\n", + " },\n", + " }\n", + "\n", + " training_pipeline = {\n", + " \"display_name\": pipeline_name,\n", + " \"training_task_definition\": schema,\n", + " \"training_task_inputs\": task,\n", + " \"input_data_config\": input_config,\n", + " \"model_to_upload\": {\"display_name\": model_name},\n", + " }\n", + "\n", + " try:\n", + " pipeline = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " )\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "task_requirements:automl,iod" + }, + "source": [ + "### Construct the task requirements\n", + "\n", + "Next, construct the task requirements. Unlike other parameters which take a Python (JSON-like) dictionary, the `task` field takes a Google protobuf Struct, which is very similar to a Python dictionary. Use the `json_format.ParseDict` method for the conversion.\n", + "\n", + "The minimal fields you need to specify are:\n", + "\n", + "- `budget_milli_node_hours`: The maximum time to budget (billed) for training the model, where 1000 = 1 hour. For image object detection, the budget must be a minimum of 20 hours.\n", + "- `model_type`: The type of deployed model:\n", + " - `CLOUD_HIGH_ACCURACY_1`: For deploying to Google Cloud and optimizing for accuracy.\n", + " - `CLOUD_LOW_LATENCY_1`: For deploying to Google Cloud and optimizing for latency (response time),\n", + " - `MOBILE_TF_HIGH_ACCURACY_1`: For deploying to the edge and optimizing for accuracy.\n", + " - `MOBILE_TF_LOW_LATENCY_1`: For deploying to the edge and optimizing for latency (response time).\n", + " - `MOBILE_TF_VERSATILE_1`: For deploying to the edge and optimizing for a trade off between latency and accuracy.\n", + "- `disable_early_stopping`: Whether True/False to let AutoML use its judgement to stop training early or train for the entire budget.\n", + "\n", + "Finally, create the pipeline by calling the helper function `create_pipeline`, which returns an instance of a training pipeline object." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_requirements:automl,iod" + }, + "outputs": [], + "source": [ + "PIPE_NAME = \"salads_pipe-\" + TIMESTAMP\n", + "MODEL_NAME = \"salads_model-\" + TIMESTAMP\n", + "\n", + "task = json_format.ParseDict(\n", + " {\n", + " \"budget_milli_node_hours\": 20000,\n", + " \"model_type\": \"CLOUD_HIGH_ACCURACY_1\",\n", + " \"disable_early_stopping\": False,\n", + " },\n", + " Value(),\n", + ")\n", + "\n", + "response = create_pipeline(PIPE_NAME, MODEL_NAME, dataset_id, TRAINING_SCHEMA, task)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 60 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "model_information" + }, + "source": [ + "## Model information\n", + "\n", + "Now that your model is trained, you can get some information on your model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:automl" + }, + "source": [ + "## Evaluate the Model resource\n", + "\n", + "Now find out how good the model service believes your model is. As part of training, some portion of the dataset was set aside as the test (holdout) data, which is used by the pipeline service to evaluate the model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "task_requirements:automl,iod" + }, + "source": [ + "### Construct the task requirements\n", + "\n", + "Next, construct the task requirements. Unlike other parameters which take a Python (JSON-like) dictionary, the `task` field takes a Google protobuf Struct, which is very similar to a Python dictionary. Use the `json_format.ParseDict` method for the conversion.\n", + "\n", + "The minimal fields you need to specify are:\n", + "\n", + "- `budget_milli_node_hours`: The maximum time to budget (billed) for training the model, where 1000 = 1 hour. For image object detection, the budget must be a minimum of 20 hours.\n", + "- `model_type`: The type of deployed model:\n", + " - `CLOUD_HIGH_ACCURACY_1`: For deploying to Google Cloud and optimizing for accuracy.\n", + " - `CLOUD_LOW_LATENCY_1`: For deploying to Google Cloud and optimizing for latency (response time),\n", + " - `MOBILE_TF_HIGH_ACCURACY_1`: For deploying to the edge and optimizing for accuracy.\n", + " - `MOBILE_TF_LOW_LATENCY_1`: For deploying to the edge and optimizing for latency (response time).\n", + " - `MOBILE_TF_VERSATILE_1`: For deploying to the edge and optimizing for a trade off between latency and accuracy.\n", + "- `disable_early_stopping`: Whether True/False to let AutoML use its judgement to stop training early or train for the entire budget.\n", + "\n", + "Finally, create the pipeline by calling the helper function `create_pipeline`, which returns an instance of a training pipeline object." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_requirements:automl,iod" + }, + "outputs": [], + "source": [ + "PIPE_NAME = \"salads_pipe-\" + TIMESTAMP\n", + "MODEL_NAME = \"salads_model-\" + TIMESTAMP\n", + "\n", + "task = json_format.ParseDict(\n", + " {\n", + " \"budget_milli_node_hours\": 20000,\n", + " \"model_type\": \"CLOUD_HIGH_ACCURACY_1\",\n", + " \"disable_early_stopping\": False,\n", + " },\n", + " Value(),\n", + ")\n", + "\n", + "response = create_pipeline(PIPE_NAME, MODEL_NAME, dataset_id, TRAINING_SCHEMA, task)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy:batch_prediction" + }, + "source": [ + "## Model deployment for batch prediction\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created for batch prediction. This differs from deploying a `Model` resource for on-demand prediction.\n", + "\n", + "For online prediction, you:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource.\n", + "\n", + "3. Make online prediction requests to the `Endpoint` resource.\n", + "\n", + "For batch-prediction, you:\n", + "\n", + "1. Create a batch prediction job.\n", + "\n", + "2. The job service will provision resources for the batch prediction request.\n", + "\n", + "3. The results of the batch prediction request are returned to the caller.\n", + "\n", + "4. The job service will unprovision the resoures for the batch prediction request." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a batch prediction request\n", + "\n", + "Now do a batch prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_items:batch_prediction" + }, + "source": [ + "### Get test item(s)\n", + "\n", + "Now do a batch prediction to your Vertex model. You will use arbitrary examples out of the dataset as a test items. Don't be concerned that the examples were likely used in training the model -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_items:automl,iod,csv" + }, + "outputs": [], + "source": [ + "test_items = !gsutil cat $IMPORT_FILE | head -n2\n", + "cols_1 = str(test_items[0]).split(\",\")\n", + "cols_2 = str(test_items[1]).split(\",\")\n", + "if len(cols_1) == 11:\n", + " test_item_1 = str(cols_1[1])\n", + " test_label_1 = str(cols_1[2])\n", + " test_item_2 = str(cols_2[1])\n", + " test_label_2 = str(cols_2[2])\n", + "else:\n", + " test_item_1 = str(cols_1[0])\n", + " test_label_1 = str(cols_1[1])\n", + " test_item_2 = str(cols_2[0])\n", + " test_label_2 = str(cols_2[1])\n", + "\n", + "print(test_item_1, test_label_1)\n", + "print(test_item_2, test_label_2)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "copy_test_items:batch_prediction" + }, + "source": [ + "### Copy test item(s)\n", + "\n", + "For the batch prediction, you will copy the test items over to your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copy_test_items:batch_prediction" + }, + "outputs": [], + "source": [ + "file_1 = test_item_1.split(\"/\")[-1]\n", + "file_2 = test_item_2.split(\"/\")[-1]\n", + "\n", + "! gsutil cp $test_item_1 $BUCKET_NAME/$file_1\n", + "! gsutil cp $test_item_2 $BUCKET_NAME/$file_2\n", + "\n", + "test_item_1 = BUCKET_NAME + \"/\" + file_1\n", + "test_item_2 = BUCKET_NAME + \"/\" + file_2" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:automl,image" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "Now make a batch input file, which you will store in your local Cloud Storage bucket. The batch input file can be either CSV or JSONL. You will use JSONL in this tutorial. For JSONL file, you make one dictionary entry per line for each data item (instance). The dictionary contains the key/value pairs:\n", + "\n", + "- `content`: The Cloud Storage path to the image.\n", + "- `mime_type`: The content type. In our example, it is an `jpeg` file.\n", + "\n", + "For example:\n", + "\n", + " {'content': '[your-bucket]/file1.jpg', 'mime_type': 'jpeg'}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_file:automl,image" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "import tensorflow as tf\n", + "\n", + "gcs_input_uri = BUCKET_NAME + \"/test.jsonl\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " data = {\"content\": test_item_1, \"mime_type\": \"image/jpeg\"}\n", + " f.write(json.dumps(data) + \"\\n\")\n", + " data = {\"content\": test_item_2, \"mime_type\": \"image/jpeg\"}\n", + " f.write(json.dumps(data) + \"\\n\")\n", + "\n", + "print(gcs_input_uri)\n", + "! gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your batch prediction requests:\n", + "\n", + "- Single Instance: The batch prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The batch prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and batch prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The batch prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_request:automl,iod" + }, + "source": [ + "### Make batch prediction request\n", + "\n", + "Now that your batch of two test items is ready, let's do the batch request. Use this helper function `create_batch_prediction_job`, with the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the prediction job.\n", + "- `model_name`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `gcs_source_uri`: The Cloud Storage path to the input file -- which you created above.\n", + "- `gcs_destination_output_uri_prefix`: The Cloud Storage path that the service will write the predictions to.\n", + "- `parameters`: Additional filtering parameters for serving prediction results.\n", + "\n", + "The helper function calls the job client service's `create_batch_prediction_job` metho, with the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for Dataset, Model and Pipeline resources.\n", + "- `batch_prediction_job`: The specification for the batch prediction job.\n", + "\n", + "Let's now dive into the specification for the `batch_prediction_job`:\n", + "\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `dedicated_resources`: The compute resources to provision for the batch prediction job.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `starting_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "- `model_parameters`: Additional filtering parameters for serving prediction results.\n", + " - `confidence_threshold`: The threshold for returning predictions. Must be between 0 and 1.\n", + " - `max_predictions`: The maximum number of predictions to return per object, sorted by confidence.\n", + "- `input_config`: The input source and format type for the instances to predict.\n", + " - `instances_format`: The format of the batch prediction request file: `csv` or `jsonl`.\n", + " - `gcs_source`: A list of one or more Cloud Storage paths to your batch prediction requests.\n", + "- `output_config`: The output destination and format for the predictions.\n", + " - `prediction_format`: The format of the batch prediction response file: `jsonl` only supported.\n", + " - `gcs_destination`: The output destination for the predictions.\n", + "\n", + "You might ask, how does confidence_threshold affect the model accuracy? The threshold won't change the accuracy. What it changes is recall and precision.\n", + "\n", + " - Precision: The higher the precision the more likely what is predicted is the correct prediction, but return fewer predictions. Increasing the confidence threshold increases precision.\n", + " - Recall: The higher the recall the more likely a correct prediction is returned in the result, but return more prediction with incorrect prediction. Decreasing the confidence threshold increases recall.\n", + "\n", + "In this example, you will predict for precision. You set the confidence threshold to 0.5 and the maximum number of predictions for a classification to two. Since, all the confidence values across the classes must add up to one, there are only two possible outcomes:\n", + "\n", + " 1. There is a tie, both 0.5, and returns two predictions.\n", + " 2. One value is above 0.5 and the rest are below 0.5, and returns one prediction.\n", + "\n", + "This call is an asychronous operation. You will print from the response object a few select fields, including:\n", + "\n", + "- `name`: The Vertex fully qualified identifier assigned to the batch prediction job.\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the Model resource.\n", + "- `generate_explanations`: Whether True/False explanations were provided with the predictions (explainability).\n", + "- `state`: The state of the prediction job (pending, running, etc).\n", + "\n", + "Since this call will take a few moments to execute, you will likely get `JobState.JOB_STATE_PENDING` for `state`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_request:automl,iod" + }, + "outputs": [], + "source": [ + "BATCH_MODEL = \"salads_batch-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_batch_prediction_job(\n", + " display_name,\n", + " model_name,\n", + " gcs_source_uri,\n", + " gcs_destination_output_uri_prefix,\n", + " parameters=None,\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " batch_prediction_job = {\n", + " \"display_name\": display_name,\n", + " # Format: 'projects/{project}/locations/{location}/models/{model_id}'\n", + " \"model\": model_name,\n", + " \"model_parameters\": json_format.ParseDict(parameters, Value()),\n", + " \"input_config\": {\n", + " \"instances_format\": IN_FORMAT,\n", + " \"gcs_source\": {\"uris\": [gcs_source_uri]},\n", + " },\n", + " \"output_config\": {\n", + " \"predictions_format\": OUT_FORMAT,\n", + " \"gcs_destination\": {\"output_uri_prefix\": gcs_destination_output_uri_prefix},\n", + " },\n", + " \"dedicated_resources\": {\n", + " \"machine_spec\": machine_spec,\n", + " \"starting_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " },\n", + " }\n", + " response = clients[\"job\"].create_batch_prediction_job(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + " )\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try:\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", response.labels)\n", + " return response\n", + "\n", + "\n", + "IN_FORMAT = \"jsonl\"\n", + "OUT_FORMAT = \"jsonl\" # [jsonl]\n", + "\n", + "response = create_batch_prediction_job(\n", + " BATCH_MODEL,\n", + " model_to_deploy_id,\n", + " gcs_input_uri,\n", + " BUCKET_NAME,\n", + " {\"confidenceThreshold\": 0.5, \"maxPredictions\": 2},\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batch_job_id:response" + }, + "source": [ + "Now get the unique identifier for the batch prediction job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batch_job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the batch job\n", + "batch_job_id = response.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split(\"/\")[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_batch_prediction_job" + }, + "source": [ + "### Get information on a batch prediction job\n", + "\n", + "Use this helper function `get_batch_prediction_job`, with the following paramter:\n", + "\n", + "- `job_name`: The Vertex fully qualified identifier for the batch prediction job.\n", + "\n", + "The helper function calls the job client service's `get_batch_prediction_job` method, with the following paramter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the batch prediction job. In this tutorial, you will pass it the Vertex fully qualified identifier for your batch prediction job -- `batch_job_id`\n", + "\n", + "The helper function will return the Cloud Storage path to where the predictions are stored -- `gcs_destination`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_batch_prediction_job" + }, + "outputs": [], + "source": [ + "def get_batch_prediction_job(job_name, silent=False):\n", + " response = clients[\"job\"].get_batch_prediction_job(name=job_name)\n", + " if silent:\n", + " return response.output_config.gcs_destination.output_uri_prefix, response.state\n", + "\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try: # not all data types support explanations\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" error:\", response.error)\n", + " gcs_destination = response.output_config.gcs_destination\n", + " print(\" gcs_destination\")\n", + " print(\" output_uri_prefix:\", gcs_destination.output_uri_prefix)\n", + " return gcs_destination.output_uri_prefix, response.state\n", + "\n", + "\n", + "predictions, state = get_batch_prediction_job(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_the_predictions:automl,iod" + }, + "source": [ + "### Get the predictions\n", + "\n", + "When the batch prediction is done processing, the job state will be `JOB_STATE_SUCCEEDED`.\n", + "\n", + "Finally you view the predictions stored at the Cloud Storage path you set as output. The predictions will be in a JSONL format, which you indicated at the time you made the batch prediction job, under a subfolder starting with the name `prediction`, and under that folder will be a file called `predictions*.jsonl`.\n", + "\n", + "Now display (cat) the contents. You will see multiple JSON objects, one for each prediction.\n", + "\n", + "The first field `ID` is the image file you did the prediction on, and the second field `annotations` is the prediction, which is further broken down into:\n", + "\n", + "- `annotations`: The detected objects.\n", + " - `display_name`: The corresponding class name.\n", + " - `score`: The percent of confidence between 0 and 1.\n", + " - `bounding_box`: The corresponding bounding box." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_the_predictions:automl,image" + }, + "outputs": [], + "source": [ + "def get_latest_predictions(gcs_out_dir):\n", + " \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n", + " folders = !gsutil ls $gcs_out_dir\n", + " latest = \"\"\n", + " for folder in folders:\n", + " subfolder = folder.split(\"/\")[-2]\n", + " if subfolder.startswith(\"prediction-\"):\n", + " if subfolder > latest:\n", + " latest = folder[:-1]\n", + " return latest\n", + "\n", + "\n", + "while True:\n", + " predictions, state = get_batch_prediction_job(batch_job_id, True)\n", + " if state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", state)\n", + " if state == aip.JobState.JOB_STATE_FAILED:\n", + " raise Exception(\"Batch Job Failed\")\n", + " else:\n", + " folder = get_latest_predictions(predictions)\n", + " ! gsutil ls $folder/prediction*.jsonl\n", + "\n", + " ! gsutil cat $folder/prediction*.jsonl\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_automl_image_object_detection_batch.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/automl/showcase_automl_image_object_detection_export_edge.ipynb b/notebooks/community/gapic/automl/showcase_automl_image_object_detection_export_edge.ipynb new file mode 100644 index 000000000..1c7be6048 --- /dev/null +++ b/notebooks/community/gapic/automl/showcase_automl_image_object_detection_export_edge.ipynb @@ -0,0 +1,1471 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: AutoML image object detection model for export to edge\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl,export_edge" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to create image object detection models to export as an Edge model using Google Cloud's AutoML." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:salads,iod" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the Salads category of the [OpenImages dataset](https://www.tensorflow.org/datasets/catalog/open_images_v4) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). This dataset does not require any feature engineering. The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts the bounding box locations and corresponding type of salad items in an image from a class of five items: salad, seafood, tomato, baked goods, or cheese." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,export_edge" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create a AutoML image object detection model from a Python script using the Vertex client library, and then export the model as an Edge model in TFLite format. You can alternatively create models with AutoML using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- Export the `Edge` model from the `Model` resource to Cloud Storage.\n", + "- Download the model locally.\n", + "- Make a local prediction." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:export_edge" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for exporting the trained model. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Set constants unique to AutoML datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the `Dataset` resource service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the `Dataset` resource service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:iod" + }, + "outputs": [], + "source": [ + "# Image Dataset type\n", + "DATA_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/metadata/image_1.0.0.yaml\"\n", + "# Image Labeling type\n", + "LABEL_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/ioformat/image_bounding_box_io_format_1.0.0.yaml\"\n", + "# Image Training task\n", + "TRAINING_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_image_object_detection_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML image object detection model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:automl,export_edge" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:automl,export_edge" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "\n", + "def create_dataset(name, schema, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " dataset = aip.Dataset(\n", + " display_name=name, metadata_schema_uri=schema, labels=labels\n", + " )\n", + "\n", + " operation = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"salads-\" + TIMESTAMP, DATA_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:image,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for images has some requirements for your data:\n", + "\n", + "- Images must be stored in a Cloud Storage bucket.\n", + "- Each image file must be in an image format (PNG, JPEG, BMP, ...).\n", + "- There must be an index file stored in your Cloud Storage bucket that contains the path and label for each image.\n", + "- The index file must be either CSV or JSONL." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:iod,u_dataset,csv" + }, + "source": [ + "#### CSV\n", + "\n", + "For image object detection, the CSV index file has the requirements:\n", + "\n", + "- No heading.\n", + "- First column is the Cloud Storage path to the image.\n", + "- Second column is the label.\n", + "- Third/Fourth columns are the upper left corner of bounding box. Coordinates are normalized, between 0 and 1.\n", + "- Fifth/Sixth/Seventh columns are not used and should be 0.\n", + "- Eighth/Ninth columns are the lower right corner of the bounding box." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:salads,csv,iod" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://cloud-samples-data/vision/salads.csv\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:csv" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Salads dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:csv" + }, + "outputs": [], + "source": [ + "if \"IMPORT_FILES\" in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_data" + }, + "source": [ + "### Import data\n", + "\n", + "Now, import the data into your Vertex Dataset resource. Use this helper function `import_data` to import the data. The function does the following:\n", + "\n", + "- Uses the `Dataset` client.\n", + "- Calls the client method `import_data`, with the following parameters:\n", + " - `name`: The human readable name you give to the `Dataset` resource (e.g., salads).\n", + " - `import_configs`: The import configuration.\n", + "\n", + "- `import_configs`: A Python list containing a dictionary, with the key/value entries:\n", + " - `gcs_sources`: A list of URIs to the paths of the one or more index files.\n", + " - `import_schema_uri`: The schema identifying the labeling type.\n", + "\n", + "The `import_data()` method returns a long running `operation` object. This will take a few minutes to complete. If you are in a live tutorial, this would be a good time to ask questions, or take a personal break." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_data" + }, + "outputs": [], + "source": [ + "def import_data(dataset, gcs_sources, schema):\n", + " config = [{\"gcs_source\": {\"uris\": gcs_sources}, \"import_schema_uri\": schema}]\n", + " print(\"dataset:\", dataset_id)\n", + " start_time = time.time()\n", + " try:\n", + " operation = clients[\"dataset\"].import_data(\n", + " name=dataset_id, import_configs=config\n", + " )\n", + " print(\"Long running operation:\", operation.operation.name)\n", + "\n", + " result = operation.result()\n", + " print(\"result:\", result)\n", + " print(\"time:\", int(time.time() - start_time), \"secs\")\n", + " print(\"error:\", operation.exception())\n", + " print(\"meta :\", operation.metadata)\n", + " print(\n", + " \"after: running:\",\n", + " operation.running(),\n", + " \"done:\",\n", + " operation.done(),\n", + " \"cancelled:\",\n", + " operation.cancelled(),\n", + " )\n", + "\n", + " return operation\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "import_data(dataset_id, [IMPORT_FILE], LABEL_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML image object detection model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_pipeline:automl" + }, + "source": [ + "### Create a training pipeline\n", + "\n", + "You may ask, what do we use a pipeline for? You typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `pipeline_name`: A human readable name for the pipeline job.\n", + "- `model_name`: A human readable name for the model.\n", + "- `dataset`: The Vertex fully qualified dataset identifier.\n", + "- `schema`: The dataset labeling (annotation) training schema.\n", + "- `task`: A dictionary describing the requirements for the training job.\n", + "\n", + "The helper function calls the `Pipeline` client service'smethod `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "Let's look now deeper into the *minimal* requirements for constructing a `training_pipeline` specification:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The dataset labeling (annotation) training schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A human readable name for the model.\n", + "- `input_data_config`: The dataset specification.\n", + " - `dataset_id`: The Vertex dataset identifier only (non-fully qualified) -- this is the last part of the fully-qualified identifier.\n", + " - `fraction_split`: If specified, the percentages of the dataset to use for training, test and validation. Otherwise, the percentages are automatically selected by AutoML." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_pipeline:automl" + }, + "outputs": [], + "source": [ + "def create_pipeline(pipeline_name, model_name, dataset, schema, task):\n", + "\n", + " dataset_id = dataset.split(\"/\")[-1]\n", + "\n", + " input_config = {\n", + " \"dataset_id\": dataset_id,\n", + " \"fraction_split\": {\n", + " \"training_fraction\": 0.8,\n", + " \"validation_fraction\": 0.1,\n", + " \"test_fraction\": 0.1,\n", + " },\n", + " }\n", + "\n", + " training_pipeline = {\n", + " \"display_name\": pipeline_name,\n", + " \"training_task_definition\": schema,\n", + " \"training_task_inputs\": task,\n", + " \"input_data_config\": input_config,\n", + " \"model_to_upload\": {\"display_name\": model_name},\n", + " }\n", + "\n", + " try:\n", + " pipeline = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " )\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "task_requirements:automl,iod,edge" + }, + "source": [ + "### Construct the task requirements\n", + "\n", + "Next, construct the task requirements. Unlike other parameters which take a Python (JSON-like) dictionary, the `task` field takes a Google protobuf Struct, which is very similar to a Python dictionary. Use the `json_format.ParseDict` method for the conversion.\n", + "\n", + "The minimal fields you need to specify are:\n", + "\n", + "- `budget_milli_node_hours`: The maximum time to budget (billed) for training the model, where 1000 = 1 hour. For image object detection, the budget must be a minimum of 20 hours.\n", + "- `model_type`: The type of deployed model:\n", + " - `CLOUD_HIGH_ACCURACY_1`: For deploying to Google Cloud and optimizing for accuracy.\n", + " - `CLOUD_LOW_LATENCY_1`: For deploying to Google Cloud and optimizing for latency (response time),\n", + " - `MOBILE_TF_HIGH_ACCURACY_1`: For deploying to the edge and optimizing for accuracy.\n", + " - `MOBILE_TF_LOW_LATENCY_1`: For deploying to the edge and optimizing for latency (response time).\n", + " - `MOBILE_TF_VERSATILE_1`: For deploying to the edge and optimizing for a trade off between latency and accuracy.\n", + "- `disable_early_stopping`: Whether True/False to let AutoML use its judgement to stop training early or train for the entire budget.\n", + "\n", + "Finally, create the pipeline by calling the helper function `create_pipeline`, which returns an instance of a training pipeline object." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_requirements:automl,iod,edge" + }, + "outputs": [], + "source": [ + "PIPE_NAME = \"salads_pipe-\" + TIMESTAMP\n", + "MODEL_NAME = \"salads_model-\" + TIMESTAMP\n", + "\n", + "task = json_format.ParseDict(\n", + " {\n", + " \"budget_milli_node_hours\": 20000,\n", + " \"model_type\": \"MOBILE_TF_LOW_LATENCY_1\",\n", + " \"disable_early_stopping\": False,\n", + " },\n", + " Value(),\n", + ")\n", + "\n", + "response = create_pipeline(PIPE_NAME, MODEL_NAME, dataset_id, TRAINING_SCHEMA, task)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 30 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "model_information" + }, + "source": [ + "## Model information\n", + "\n", + "Now that your model is trained, you can get some information on your model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:automl" + }, + "source": [ + "## Evaluate the Model resource\n", + "\n", + "Now find out how good the model service believes your model is. As part of training, some portion of the dataset was set aside as the test (holdout) data, which is used by the pipeline service to evaluate the model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "list_model_evaluations:automl,iod" + }, + "source": [ + "### List evaluations for all slices\n", + "\n", + "Use this helper function `list_model_evaluations`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified model identifier for the `Model` resource.\n", + "\n", + "This helper function uses the model client service's `list_model_evaluations` method, which takes the same parameter. The response object from the call is a list, where each element is an evaluation metric.\n", + "\n", + "For each evaluation -- you probably only have one, we then print all the key names for each metric in the evaluation, and for a small set (`evaluatedBoundingBoxCount` and `boundingBoxMeanAveragePrecision`) you will print the result." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "list_model_evaluations:automl,iod" + }, + "outputs": [], + "source": [ + "def list_model_evaluations(name):\n", + " response = clients[\"model\"].list_model_evaluations(parent=name)\n", + " for evaluation in response:\n", + " print(\"model_evaluation\")\n", + " print(\" name:\", evaluation.name)\n", + " print(\" metrics_schema_uri:\", evaluation.metrics_schema_uri)\n", + " metrics = json_format.MessageToDict(evaluation._pb.metrics)\n", + " for metric in metrics.keys():\n", + " print(metric)\n", + " print(\"evaluatedBoundingBoxCount\", metrics[\"evaluatedBoundingBoxCount\"])\n", + " print(\n", + " \"boundingBoxMeanAveragePrecision\",\n", + " metrics[\"boundingBoxMeanAveragePrecision\"],\n", + " )\n", + "\n", + " return evaluation.name\n", + "\n", + "\n", + "last_evaluation = list_model_evaluations(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "export_model:export_edge" + }, + "source": [ + "## Export as Edge model\n", + "\n", + "You can export an AutoML image object detection model as an Edge model which you can then custom deploy to an edge device, such as a mobile phone or IoT device, or download locally. Use this helper function `export_model` to export the model to Google Cloud, which takes the following parameters:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `format`: The format to save the model format as.\n", + "- `gcs_dest`: The Cloud Storage location to store the SavedFormat model artifacts to.\n", + "\n", + "This function calls the `Model` client service's method `export_model`, with the following parameters:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `output_config`: The destination information for the exported model.\n", + " - `artifact_destination.output_uri_prefix`: The Cloud Storage location to store the SavedFormat model artifacts to.\n", + " - `export_format_id`: The format to save the model format as. For AutoML image object detection:\n", + " - `tf-saved-model`: TensorFlow SavedFormat for deployment to a container.\n", + " - `tflite`: TensorFlow Lite for deployment to an edge or mobile device.\n", + " - `edgetpu-tflite`: TensorFlow Lite for TPU\n", + " - `tf-js`: TensorFlow for web client\n", + " - `coral-ml`: for Coral devices\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is exported." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "export_model:export_edge" + }, + "outputs": [], + "source": [ + "MODEL_DIR = BUCKET_NAME + \"/\" + \"salads\"\n", + "\n", + "\n", + "def export_model(name, format, gcs_dest):\n", + " output_config = {\n", + " \"artifact_destination\": {\"output_uri_prefix\": gcs_dest},\n", + " \"export_format_id\": format,\n", + " }\n", + " response = clients[\"model\"].export_model(name=name, output_config=output_config)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result(timeout=1800)\n", + " metadata = response.operation.metadata\n", + " artifact_uri = str(metadata.value).split(\"\\\\\")[-1][4:-1]\n", + " print(\"Artifact Uri\", artifact_uri)\n", + " return artifact_uri\n", + "\n", + "\n", + "model_package = export_model(model_to_deploy_id, \"tflite\", MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "download_model_artifacts:tflite" + }, + "source": [ + "#### Download the TFLite model artifacts\n", + "\n", + "Now that you have an exported TFLite version of your model, you can test the exported model locally, but first downloading it from Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "download_model_artifacts:tflite" + }, + "outputs": [], + "source": [ + "! gsutil ls $model_package\n", + "# Download the model artifacts\n", + "! gsutil cp -r $model_package tflite\n", + "\n", + "tflite_path = \"tflite/model.tflite\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instantiate_tflite_interpreter" + }, + "source": [ + "#### Instantiate a TFLite interpreter\n", + "\n", + "The TFLite version of the model is not a TensorFlow SavedModel format. You cannot directly use methods like predict(). Instead, one uses the TFLite interpreter. You must first setup the interpreter for the TFLite model as follows:\n", + "\n", + "- Instantiate an TFLite interpreter for the TFLite model.\n", + "- Instruct the interpreter to allocate input and output tensors for the model.\n", + "- Get detail information about the models input and output tensors that will need to be known for prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instantiate_tflite_interpreter" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "interpreter = tf.lite.Interpreter(model_path=tflite_path)\n", + "interpreter.allocate_tensors()\n", + "\n", + "input_details = interpreter.get_input_details()\n", + "output_details = interpreter.get_output_details()\n", + "input_shape = input_details[0][\"shape\"]\n", + "\n", + "print(\"input tensor shape\", input_shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_item" + }, + "source": [ + "### Get test item\n", + "\n", + "You will use an arbitrary example out of the dataset as a test item. Don't be concerned that the example was likely used in training the model -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:image,224x224" + }, + "outputs": [], + "source": [ + "test_items = ! gsutil cat $IMPORT_FILE | head -n1\n", + "test_item = test_items[0].split(\",\")[0]\n", + "\n", + "with tf.io.gfile.GFile(test_item, \"rb\") as f:\n", + " content = f.read()\n", + "test_image = tf.io.decode_jpeg(content)\n", + "print(\"test image shape\", test_image.shape)\n", + "\n", + "test_image = tf.image.resize(test_image, (224, 224))\n", + "print(\"test image shape\", test_image.shape, test_image.dtype)\n", + "\n", + "test_image = tf.cast(test_image, dtype=tf.uint8).numpy()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "invoke_tflite_interpreter" + }, + "source": [ + "#### Make a prediction with TFLite model\n", + "\n", + "Finally, you do a prediction using your TFLite model, as follows:\n", + "\n", + "- Convert the test image into a batch of a single image (`np.expand_dims`)\n", + "- Set the input tensor for the interpreter to your batch of a single image (`data`).\n", + "- Invoke the interpreter.\n", + "- Retrieve the softmax probabilities for the prediction (`get_tensor`).\n", + "- Determine which label had the highest probability (`np.argmax`)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "invoke_tflite_interpreter" + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "\n", + "data = np.expand_dims(test_image, axis=0)\n", + "\n", + "interpreter.set_tensor(input_details[0][\"index\"], data)\n", + "\n", + "interpreter.invoke()\n", + "\n", + "softmax = interpreter.get_tensor(output_details[0][\"index\"])\n", + "\n", + "label = np.argmax(softmax)\n", + "\n", + "print(label)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_automl_image_object_detection_export_edge.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/automl/showcase_automl_image_object_detection_online.ipynb b/notebooks/community/gapic/automl/showcase_automl_image_object_detection_online.ipynb new file mode 100644 index 000000000..0aaa3199d --- /dev/null +++ b/notebooks/community/gapic/automl/showcase_automl_image_object_detection_online.ipynb @@ -0,0 +1,1593 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: AutoML image object detection model for online prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to create image object detection models and do online prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:salads,iod" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the Salads category of the [OpenImages dataset](https://www.tensorflow.org/datasets/catalog/open_images_v4) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). This dataset does not require any feature engineering. The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts the bounding box locations and corresponding type of salad items in an image from a class of five items: salad, seafood, tomato, baked goods, or cheese." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,online_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML image object detection model and deploy for online prediction from a Python script using the Vertex client library. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model`." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Set constants unique to AutoML datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the `Dataset` resource service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the `Dataset` resource service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:iod" + }, + "outputs": [], + "source": [ + "# Image Dataset type\n", + "DATA_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/metadata/image_1.0.0.yaml\"\n", + "# Image Labeling type\n", + "LABEL_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/ioformat/image_bounding_box_io_format_1.0.0.yaml\"\n", + "# Image Training task\n", + "TRAINING_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_image_object_detection_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML image object detection model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:automl,online_prediction" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training.\n", + "- Endpoint Service for deployment.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:automl,online_prediction" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "\n", + "def create_dataset(name, schema, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " dataset = aip.Dataset(\n", + " display_name=name, metadata_schema_uri=schema, labels=labels\n", + " )\n", + "\n", + " operation = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"salads-\" + TIMESTAMP, DATA_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:image,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for images has some requirements for your data:\n", + "\n", + "- Images must be stored in a Cloud Storage bucket.\n", + "- Each image file must be in an image format (PNG, JPEG, BMP, ...).\n", + "- There must be an index file stored in your Cloud Storage bucket that contains the path and label for each image.\n", + "- The index file must be either CSV or JSONL." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:iod,u_dataset,csv" + }, + "source": [ + "#### CSV\n", + "\n", + "For image object detection, the CSV index file has the requirements:\n", + "\n", + "- No heading.\n", + "- First column is the Cloud Storage path to the image.\n", + "- Second column is the label.\n", + "- Third/Fourth columns are the upper left corner of bounding box. Coordinates are normalized, between 0 and 1.\n", + "- Fifth/Sixth/Seventh columns are not used and should be 0.\n", + "- Eighth/Ninth columns are the lower right corner of the bounding box." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:salads,csv,iod" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://cloud-samples-data/vision/salads.csv\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:csv" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Salads dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:csv" + }, + "outputs": [], + "source": [ + "if \"IMPORT_FILES\" in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_data" + }, + "source": [ + "### Import data\n", + "\n", + "Now, import the data into your Vertex Dataset resource. Use this helper function `import_data` to import the data. The function does the following:\n", + "\n", + "- Uses the `Dataset` client.\n", + "- Calls the client method `import_data`, with the following parameters:\n", + " - `name`: The human readable name you give to the `Dataset` resource (e.g., salads).\n", + " - `import_configs`: The import configuration.\n", + "\n", + "- `import_configs`: A Python list containing a dictionary, with the key/value entries:\n", + " - `gcs_sources`: A list of URIs to the paths of the one or more index files.\n", + " - `import_schema_uri`: The schema identifying the labeling type.\n", + "\n", + "The `import_data()` method returns a long running `operation` object. This will take a few minutes to complete. If you are in a live tutorial, this would be a good time to ask questions, or take a personal break." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_data" + }, + "outputs": [], + "source": [ + "def import_data(dataset, gcs_sources, schema):\n", + " config = [{\"gcs_source\": {\"uris\": gcs_sources}, \"import_schema_uri\": schema}]\n", + " print(\"dataset:\", dataset_id)\n", + " start_time = time.time()\n", + " try:\n", + " operation = clients[\"dataset\"].import_data(\n", + " name=dataset_id, import_configs=config\n", + " )\n", + " print(\"Long running operation:\", operation.operation.name)\n", + "\n", + " result = operation.result()\n", + " print(\"result:\", result)\n", + " print(\"time:\", int(time.time() - start_time), \"secs\")\n", + " print(\"error:\", operation.exception())\n", + " print(\"meta :\", operation.metadata)\n", + " print(\n", + " \"after: running:\",\n", + " operation.running(),\n", + " \"done:\",\n", + " operation.done(),\n", + " \"cancelled:\",\n", + " operation.cancelled(),\n", + " )\n", + "\n", + " return operation\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "import_data(dataset_id, [IMPORT_FILE], LABEL_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML image object detection model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_pipeline:automl" + }, + "source": [ + "### Create a training pipeline\n", + "\n", + "You may ask, what do we use a pipeline for? You typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `pipeline_name`: A human readable name for the pipeline job.\n", + "- `model_name`: A human readable name for the model.\n", + "- `dataset`: The Vertex fully qualified dataset identifier.\n", + "- `schema`: The dataset labeling (annotation) training schema.\n", + "- `task`: A dictionary describing the requirements for the training job.\n", + "\n", + "The helper function calls the `Pipeline` client service'smethod `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "Let's look now deeper into the *minimal* requirements for constructing a `training_pipeline` specification:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The dataset labeling (annotation) training schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A human readable name for the model.\n", + "- `input_data_config`: The dataset specification.\n", + " - `dataset_id`: The Vertex dataset identifier only (non-fully qualified) -- this is the last part of the fully-qualified identifier.\n", + " - `fraction_split`: If specified, the percentages of the dataset to use for training, test and validation. Otherwise, the percentages are automatically selected by AutoML." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_pipeline:automl" + }, + "outputs": [], + "source": [ + "def create_pipeline(pipeline_name, model_name, dataset, schema, task):\n", + "\n", + " dataset_id = dataset.split(\"/\")[-1]\n", + "\n", + " input_config = {\n", + " \"dataset_id\": dataset_id,\n", + " \"fraction_split\": {\n", + " \"training_fraction\": 0.8,\n", + " \"validation_fraction\": 0.1,\n", + " \"test_fraction\": 0.1,\n", + " },\n", + " }\n", + "\n", + " training_pipeline = {\n", + " \"display_name\": pipeline_name,\n", + " \"training_task_definition\": schema,\n", + " \"training_task_inputs\": task,\n", + " \"input_data_config\": input_config,\n", + " \"model_to_upload\": {\"display_name\": model_name},\n", + " }\n", + "\n", + " try:\n", + " pipeline = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " )\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "task_requirements:automl,iod" + }, + "source": [ + "### Construct the task requirements\n", + "\n", + "Next, construct the task requirements. Unlike other parameters which take a Python (JSON-like) dictionary, the `task` field takes a Google protobuf Struct, which is very similar to a Python dictionary. Use the `json_format.ParseDict` method for the conversion.\n", + "\n", + "The minimal fields you need to specify are:\n", + "\n", + "- `budget_milli_node_hours`: The maximum time to budget (billed) for training the model, where 1000 = 1 hour. For image object detection, the budget must be a minimum of 20 hours.\n", + "- `model_type`: The type of deployed model:\n", + " - `CLOUD_HIGH_ACCURACY_1`: For deploying to Google Cloud and optimizing for accuracy.\n", + " - `CLOUD_LOW_LATENCY_1`: For deploying to Google Cloud and optimizing for latency (response time),\n", + " - `MOBILE_TF_HIGH_ACCURACY_1`: For deploying to the edge and optimizing for accuracy.\n", + " - `MOBILE_TF_LOW_LATENCY_1`: For deploying to the edge and optimizing for latency (response time).\n", + " - `MOBILE_TF_VERSATILE_1`: For deploying to the edge and optimizing for a trade off between latency and accuracy.\n", + "- `disable_early_stopping`: Whether True/False to let AutoML use its judgement to stop training early or train for the entire budget.\n", + "\n", + "Finally, create the pipeline by calling the helper function `create_pipeline`, which returns an instance of a training pipeline object." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_requirements:automl,iod" + }, + "outputs": [], + "source": [ + "PIPE_NAME = \"salads_pipe-\" + TIMESTAMP\n", + "MODEL_NAME = \"salads_model-\" + TIMESTAMP\n", + "\n", + "task = json_format.ParseDict(\n", + " {\n", + " \"budget_milli_node_hours\": 20000,\n", + " \"model_type\": \"CLOUD_HIGH_ACCURACY_1\",\n", + " \"disable_early_stopping\": False,\n", + " },\n", + " Value(),\n", + ")\n", + "\n", + "response = create_pipeline(PIPE_NAME, MODEL_NAME, dataset_id, TRAINING_SCHEMA, task)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 60 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "model_information" + }, + "source": [ + "## Model information\n", + "\n", + "Now that your model is trained, you can get some information on your model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:automl" + }, + "source": [ + "## Evaluate the Model resource\n", + "\n", + "Now find out how good the model service believes your model is. As part of training, some portion of the dataset was set aside as the test (holdout) data, which is used by the pipeline service to evaluate the model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "list_model_evaluations:automl,iod" + }, + "source": [ + "### List evaluations for all slices\n", + "\n", + "Use this helper function `list_model_evaluations`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified model identifier for the `Model` resource.\n", + "\n", + "This helper function uses the model client service's `list_model_evaluations` method, which takes the same parameter. The response object from the call is a list, where each element is an evaluation metric.\n", + "\n", + "For each evaluation -- you probably only have one, we then print all the key names for each metric in the evaluation, and for a small set (`evaluatedBoundingBoxCount` and `boundingBoxMeanAveragePrecision`) you will print the result." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "list_model_evaluations:automl,iod" + }, + "outputs": [], + "source": [ + "def list_model_evaluations(name):\n", + " response = clients[\"model\"].list_model_evaluations(parent=name)\n", + " for evaluation in response:\n", + " print(\"model_evaluation\")\n", + " print(\" name:\", evaluation.name)\n", + " print(\" metrics_schema_uri:\", evaluation.metrics_schema_uri)\n", + " metrics = json_format.MessageToDict(evaluation._pb.metrics)\n", + " for metric in metrics.keys():\n", + " print(metric)\n", + " print(\"evaluatedBoundingBoxCount\", metrics[\"evaluatedBoundingBoxCount\"])\n", + " print(\n", + " \"boundingBoxMeanAveragePrecision\",\n", + " metrics[\"boundingBoxMeanAveragePrecision\"],\n", + " )\n", + "\n", + " return evaluation.name\n", + "\n", + "\n", + "last_evaluation = list_model_evaluations(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:automl" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created with AutoML. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"salads_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:automatic" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `automatic_resources`: This refers to how many redundant compute instances (replicas). For this example, we set it to one (no replication).\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:automatic" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"salads_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"automatic_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " },\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_item" + }, + "source": [ + "### Get test item\n", + "\n", + "You will use an arbitrary example out of the dataset as a test item. Don't be concerned that the example was likely used in training the model -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:automl,iod,csv" + }, + "outputs": [], + "source": [ + "test_items = !gsutil cat $IMPORT_FILE | head -n1\n", + "cols = str(test_items[0]).split(\",\")\n", + "if len(cols) == 11:\n", + " test_item = str(cols[1])\n", + " test_label = str(cols[2])\n", + "else:\n", + " test_item = str(cols[0])\n", + " test_label = str(cols[1])\n", + "\n", + "print(test_item, test_label)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "predict_item:automl,iod" + }, + "source": [ + "### Make a prediction\n", + "\n", + "Now you have a test item. Use this helper function `predict_item`, which takes the following parameters:\n", + "\n", + "- `filename`: The Cloud Storage path to the test item.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `parameters_dict`: Additional filtering parameters for serving prediction results.\n", + "\n", + "This function calls the prediction client service's `predict` method with the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `instances`: A list of instances (encoded images) to predict.\n", + "- `parameters`: Additional parameters for serving.\n", + " - `confidence_threshold`: The threshold for returning predictions. Must be between 0 and 1.\n", + " - `max_predictions`: The maximum number of predictions per object to return, sorted by confidence.\n", + "\n", + "You might ask, how does confidence_threshold affect the model accuracy? The threshold won't change the accuracy. What it changes is recall and precision.\n", + "\n", + " - Precision: The higher the precision the more likely what is predicted is the correct prediction, but return fewer predictions. Increasing the confidence threshold increases precision.\n", + " - Recall: The higher the recall the more likely a correct prediction is returned in the result, but return more prediction with incorrect prediction. Decreasing the confidence threshold increases recall.\n", + "\n", + "In this example, you will predict for precision. You set the confidence threshold to 0.5 and the maximum number of predictions for an object to two. Since, all the confidence values across the classes must add up to one, there are only two possible outcomes:\n", + "\n", + " 1. There is a tie, both 0.5, and returns two predictions.\n", + " 2. One value is above 0.5 and all the rest are below 0.5, and returns one prediction.\n", + "\n", + "#### Request\n", + "\n", + "Since in this example your test item is in a Cloud Storage bucket, you will open and read the contents of the image using `tf.io.gfile.Gfile()`. To pass the test data to the prediction service, we will encode the bytes into base 64 -- This makes binary data safe from modification while it is transferred over the Internet.\n", + "\n", + "The format of each instance is:\n", + "\n", + " { 'content': { 'b64': [base64_encoded_bytes] } }\n", + "\n", + "Since the `predict()` method can take multiple items (instances), you send our single test item as a list of one test item. As a final step, you package the instances list into Google's protobuf format -- which is what we pass to the `predict()` method.\n", + "\n", + "#### Response\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding image in the request. You will see in the output for each prediction -- in our case there is just one:\n", + "\n", + "- `confidences`: Confidence level in the prediction.\n", + "- `displayNames`: The predicted label.\n", + "- `bboxes`: The bounding box for the label." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "predict_item:automl,iod" + }, + "outputs": [], + "source": [ + "import base64\n", + "\n", + "import tensorflow as tf\n", + "\n", + "\n", + "def predict_item(filename, endpoint, parameters_dict):\n", + "\n", + " parameters = json_format.ParseDict(parameters_dict, Value())\n", + " with tf.io.gfile.GFile(filename, \"rb\") as f:\n", + " content = f.read()\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [{\"content\": base64.b64encode(content).decode(\"utf-8\")}]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", dict(prediction))\n", + "\n", + "\n", + "predict_item(test_item, endpoint_id, {\"confidenceThreshold\": 0.5, \"maxPredictions\": 2})" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_automl_image_object_detection_online.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/automl/showcase_automl_image_segmentation_batch.ipynb b/notebooks/community/gapic/automl/showcase_automl_image_segmentation_batch.ipynb new file mode 100644 index 000000000..43507b7aa --- /dev/null +++ b/notebooks/community/gapic/automl/showcase_automl_image_segmentation_batch.ipynb @@ -0,0 +1,1815 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: AutoML image segmentation model for batch prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to create image segmentation models and do batch prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:isg_unknown,isg" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [TODO](https://). This dataset does not require any feature engineering. The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,batch_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML image segmentation model from a Python script, and then do a batch prediction using the Vertex client library. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Make a batch prediction.\n", + "\n", + "There is one key difference between using batch prediction and using online prediction:\n", + "\n", + "* Prediction Service: Does an on-demand prediction for the entire set of instances (i.e., one or more data items) and returns the results in real-time.\n", + "\n", + "* Batch Prediction Service: Does a queued (batch) prediction for the entire set of instances in the background and stores the results in a Cloud Storage bucket when ready." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Set constants unique to AutoML datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the `Dataset` resource service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the `Dataset` resource service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:isg" + }, + "outputs": [], + "source": [ + "# Image Dataset type\n", + "DATA_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/metadata/image_1.0.0.yaml\"\n", + "# Image Labeling type\n", + "LABEL_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/ioformat/image_segmentation_io_format_1.0.0.yaml\"\n", + "# Image Training task\n", + "TRAINING_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_image_segmentation_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:prediction" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for prediction.\n", + "\n", + "Set the variable `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:automl" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "For AutoML batch prediction, the container image for the serving binary is pre-determined by the Vertex prediction service. More specifically, the service will pick the appropriate container for the model depending on the hardware accelerator you selected." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for prediction.\n", + "\n", + "- Set the variable `DEPLOY_COMPUTE` to configure the compute resources for the VM you will use for prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML image segmentation model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:automl,batch_prediction" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training.\n", + "- Job Service for batch prediction and custom training." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:automl,batch_prediction" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"job\"] = create_job_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "\n", + "def create_dataset(name, schema, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " dataset = aip.Dataset(\n", + " display_name=name, metadata_schema_uri=schema, labels=labels\n", + " )\n", + "\n", + " operation = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"unknown-\" + TIMESTAMP, DATA_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:image,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for images has some requirements for your data:\n", + "\n", + "- Images must be stored in a Cloud Storage bucket.\n", + "- Each image file must be in an image format (PNG, JPEG, BMP, ...).\n", + "- There must be an index file stored in your Cloud Storage bucket that contains the path and label for each image.\n", + "- The index file must be either CSV or JSONL." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:isg,u_dataset,jsonl" + }, + "source": [ + "#### JSONL\n", + "\n", + "For image segmentation, the JSONL index file has the requirements:\n", + "\n", + "- Each data item is a separate JSON object, on a separate line.\n", + "- The key/value pair `image_gcs_uri` is the Cloud Storage path to the image.\n", + "- The key/value pair `category_mask_uri` is the Cloud Storage path to the mask image in PNG format.\n", + "- The key/value pair `'annotation_spec_colors'` is a list mapping mask colors to a label.\n", + " - The key/value pair pair `display_name` is the label for the pixel color mask.\n", + " - The key/value pair pair `color` are the RGB normalized pixel values (between 0 and 1) of the mask for the corresponding label.\n", + "\n", + " { 'image_gcs_uri': image, 'segmentation_annotations': { 'category_mask_uri': mask_image, 'annotation_spec_colors' : [ { 'display_name': label, 'color': {\"red\": value, \"blue\", value, \"green\": value} }, ...] }\n", + "\n", + "*Note*: The dictionary key fields may alternatively be in camelCase. For example, 'image_gcs_uri' can also be 'imageGcsUri'." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,jsonl" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the JSONL index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:isg_unknown,jsonl,isg" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://ucaip-test-us-central1/dataset/isg_data.jsonl\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:jsonl" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Unknown dataset that is stored in a public Cloud Storage bucket, using a JSONL index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of objects in a JSONL index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:jsonl" + }, + "outputs": [], + "source": [ + "if \"IMPORT_FILES\" in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_data" + }, + "source": [ + "### Import data\n", + "\n", + "Now, import the data into your Vertex Dataset resource. Use this helper function `import_data` to import the data. The function does the following:\n", + "\n", + "- Uses the `Dataset` client.\n", + "- Calls the client method `import_data`, with the following parameters:\n", + " - `name`: The human readable name you give to the `Dataset` resource (e.g., unknown).\n", + " - `import_configs`: The import configuration.\n", + "\n", + "- `import_configs`: A Python list containing a dictionary, with the key/value entries:\n", + " - `gcs_sources`: A list of URIs to the paths of the one or more index files.\n", + " - `import_schema_uri`: The schema identifying the labeling type.\n", + "\n", + "The `import_data()` method returns a long running `operation` object. This will take a few minutes to complete. If you are in a live tutorial, this would be a good time to ask questions, or take a personal break." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_data" + }, + "outputs": [], + "source": [ + "def import_data(dataset, gcs_sources, schema):\n", + " config = [{\"gcs_source\": {\"uris\": gcs_sources}, \"import_schema_uri\": schema}]\n", + " print(\"dataset:\", dataset_id)\n", + " start_time = time.time()\n", + " try:\n", + " operation = clients[\"dataset\"].import_data(\n", + " name=dataset_id, import_configs=config\n", + " )\n", + " print(\"Long running operation:\", operation.operation.name)\n", + "\n", + " result = operation.result()\n", + " print(\"result:\", result)\n", + " print(\"time:\", int(time.time() - start_time), \"secs\")\n", + " print(\"error:\", operation.exception())\n", + " print(\"meta :\", operation.metadata)\n", + " print(\n", + " \"after: running:\",\n", + " operation.running(),\n", + " \"done:\",\n", + " operation.done(),\n", + " \"cancelled:\",\n", + " operation.cancelled(),\n", + " )\n", + "\n", + " return operation\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "import_data(dataset_id, [IMPORT_FILE], LABEL_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML image segmentation model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_pipeline:automl" + }, + "source": [ + "### Create a training pipeline\n", + "\n", + "You may ask, what do we use a pipeline for? You typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `pipeline_name`: A human readable name for the pipeline job.\n", + "- `model_name`: A human readable name for the model.\n", + "- `dataset`: The Vertex fully qualified dataset identifier.\n", + "- `schema`: The dataset labeling (annotation) training schema.\n", + "- `task`: A dictionary describing the requirements for the training job.\n", + "\n", + "The helper function calls the `Pipeline` client service'smethod `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "Let's look now deeper into the *minimal* requirements for constructing a `training_pipeline` specification:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The dataset labeling (annotation) training schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A human readable name for the model.\n", + "- `input_data_config`: The dataset specification.\n", + " - `dataset_id`: The Vertex dataset identifier only (non-fully qualified) -- this is the last part of the fully-qualified identifier.\n", + " - `fraction_split`: If specified, the percentages of the dataset to use for training, test and validation. Otherwise, the percentages are automatically selected by AutoML." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_pipeline:automl" + }, + "outputs": [], + "source": [ + "def create_pipeline(pipeline_name, model_name, dataset, schema, task):\n", + "\n", + " dataset_id = dataset.split(\"/\")[-1]\n", + "\n", + " input_config = {\n", + " \"dataset_id\": dataset_id,\n", + " \"fraction_split\": {\n", + " \"training_fraction\": 0.8,\n", + " \"validation_fraction\": 0.1,\n", + " \"test_fraction\": 0.1,\n", + " },\n", + " }\n", + "\n", + " training_pipeline = {\n", + " \"display_name\": pipeline_name,\n", + " \"training_task_definition\": schema,\n", + " \"training_task_inputs\": task,\n", + " \"input_data_config\": input_config,\n", + " \"model_to_upload\": {\"display_name\": model_name},\n", + " }\n", + "\n", + " try:\n", + " pipeline = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " )\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "task_requirements:automl,isg" + }, + "source": [ + "### Construct the task requirements\n", + "\n", + "Next, construct the task requirements. Unlike other parameters which take a Python (JSON-like) dictionary, the `task` field takes a Google protobuf Struct, which is very similar to a Python dictionary. Use the `json_format.ParseDict` method for the conversion.\n", + "\n", + "The minimal fields you need to specify are:\n", + "\n", + "- `budget_milli_node_hours`: The maximum time to budget (billed) for training the model, where 1000 = 1 hour.\n", + "- `model_type`: The type of deployed model:\n", + " - `CLOUD_HIGH_ACCURACY_1`: For deploying to Google Cloud and optimizing for accuracy.\n", + " - `CLOUD_LOW_LATENCY_1`: For deploying to Google Cloud and optimizing for latency (response time),\n", + "\n", + "Finally, create the pipeline by calling the helper function `create_pipeline`, which returns an instance of a training pipeline object." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_requirements:automl,isg" + }, + "outputs": [], + "source": [ + "PIPE_NAME = \"unknown_pipe-\" + TIMESTAMP\n", + "MODEL_NAME = \"unknown_model-\" + TIMESTAMP\n", + "\n", + "task = json_format.ParseDict(\n", + " {\"budget_milli_node_hours\": 2000, \"model_type\": \"CLOUD_LOW_ACCURACY_1\"}, Value()\n", + ")\n", + "\n", + "response = create_pipeline(PIPE_NAME, MODEL_NAME, dataset_id, TRAINING_SCHEMA, task)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 30 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "model_information" + }, + "source": [ + "## Model information\n", + "\n", + "Now that your model is trained, you can get some information on your model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:automl" + }, + "source": [ + "## Evaluate the Model resource\n", + "\n", + "Now find out how good the model service believes your model is. As part of training, some portion of the dataset was set aside as the test (holdout) data, which is used by the pipeline service to evaluate the model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "list_model_evaluations:automl,isg" + }, + "source": [ + "### List evaluations for all slices\n", + "\n", + "Use this helper function `list_model_evaluations`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified model identifier for the `Model` resource.\n", + "\n", + "This helper function uses the model client service's `list_model_evaluations` method, which takes the same parameter. The response object from the call is a list, where each element is an evaluation metric.\n", + "\n", + "For each evaluation -- you probably only have one, we then print all the key names for each metric in the evaluation, and for a small set (`confidenceMetricsEntries`) you will print the result." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "list_model_evaluations:automl,isg" + }, + "outputs": [], + "source": [ + "def list_model_evaluations(name):\n", + " response = clients[\"model\"].list_model_evaluations(parent=name)\n", + " for evaluation in response:\n", + " print(\"model_evaluation\")\n", + " print(\" name:\", evaluation.name)\n", + " print(\" metrics_schema_uri:\", evaluation.metrics_schema_uri)\n", + " metrics = json_format.MessageToDict(evaluation._pb.metrics)\n", + " for metric in metrics.keys():\n", + " print(metric)\n", + " print(\"confidenceMetricsEntries\", metrics[\"confidenceMetricsEntries\"])\n", + "\n", + " return evaluation.name\n", + "\n", + "\n", + "last_evaluation = list_model_evaluations(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy:batch_prediction" + }, + "source": [ + "## Model deployment for batch prediction\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created for batch prediction. This differs from deploying a `Model` resource for on-demand prediction.\n", + "\n", + "For online prediction, you:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource.\n", + "\n", + "3. Make online prediction requests to the `Endpoint` resource.\n", + "\n", + "For batch-prediction, you:\n", + "\n", + "1. Create a batch prediction job.\n", + "\n", + "2. The job service will provision resources for the batch prediction request.\n", + "\n", + "3. The results of the batch prediction request are returned to the caller.\n", + "\n", + "4. The job service will unprovision the resoures for the batch prediction request." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a batch prediction request\n", + "\n", + "Now do a batch prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_items:batch_prediction" + }, + "source": [ + "### Get test item(s)\n", + "\n", + "Now do a batch prediction to your Vertex model. You will use arbitrary examples out of the dataset as a test items. Don't be concerned that the examples were likely used in training the model -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_items:automl,isg,jsonl" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "test_items = !gsutil cat $IMPORT_FILE | head -n2\n", + "test_data_1 = test_items[0].replace(\"'\", '\"')\n", + "test_data_1 = json.loads(test_data_1)\n", + "test_data_2 = test_items[0].replace(\"'\", '\"')\n", + "test_data_2 = json.loads(test_data_2)\n", + "try:\n", + " test_item_1 = test_data_1[\"image_gcs_uri\"]\n", + " test_label_1 = test_data_1[\"segmentation_annotation\"][\"annotation_spec_colors\"]\n", + " test_item_2 = test_data_2[\"image_gcs_uri\"]\n", + " test_label_2 = test_data_2[\"segmentation_annotation\"][\"annotation_spec_colors\"]\n", + "except:\n", + " test_item_1 = test_data_1[\"imageGcsUri\"]\n", + " test_label_1 = test_data_1[\"segmentationAnnotation\"][\"annotationSpecColors\"]\n", + " test_item_2 = test_data_2[\"imageGcsUri\"]\n", + " test_label_2 = test_data_2[\"segmentationAnnotation\"][\"annotationSpecColors\"]\n", + "\n", + "print(test_item_1, test_label_1)\n", + "print(test_item_2, test_label_2)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "copy_test_items:batch_prediction" + }, + "source": [ + "### Copy test item(s)\n", + "\n", + "For the batch prediction, you will copy the test items over to your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copy_test_items:batch_prediction" + }, + "outputs": [], + "source": [ + "file_1 = test_item_1.split(\"/\")[-1]\n", + "file_2 = test_item_2.split(\"/\")[-1]\n", + "\n", + "! gsutil cp $test_item_1 $BUCKET_NAME/$file_1\n", + "! gsutil cp $test_item_2 $BUCKET_NAME/$file_2\n", + "\n", + "test_item_1 = BUCKET_NAME + \"/\" + file_1\n", + "test_item_2 = BUCKET_NAME + \"/\" + file_2" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:automl,image" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "Now make a batch input file, which you will store in your local Cloud Storage bucket. The batch input file can be either CSV or JSONL. You will use JSONL in this tutorial. For JSONL file, you make one dictionary entry per line for each data item (instance). The dictionary contains the key/value pairs:\n", + "\n", + "- `content`: The Cloud Storage path to the image.\n", + "- `mime_type`: The content type. In our example, it is an `jpeg` file.\n", + "\n", + "For example:\n", + "\n", + " {'content': '[your-bucket]/file1.jpg', 'mime_type': 'jpeg'}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_file:automl,image" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "import tensorflow as tf\n", + "\n", + "gcs_input_uri = BUCKET_NAME + \"/test.jsonl\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " data = {\"content\": test_item_1, \"mime_type\": \"image/jpeg\"}\n", + " f.write(json.dumps(data) + \"\\n\")\n", + " data = {\"content\": test_item_2, \"mime_type\": \"image/jpeg\"}\n", + " f.write(json.dumps(data) + \"\\n\")\n", + "\n", + "print(gcs_input_uri)\n", + "! gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your batch prediction requests:\n", + "\n", + "- Single Instance: The batch prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The batch prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and batch prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The batch prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_request:automl,isg" + }, + "source": [ + "### Make batch prediction request\n", + "\n", + "Now that your batch of two test items is ready, let's do the batch request. Use this helper function `create_batch_prediction_job`, with the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the prediction job.\n", + "- `model_name`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `gcs_source_uri`: The Cloud Storage path to the input file -- which you created above.\n", + "- `gcs_destination_output_uri_prefix`: The Cloud Storage path that the service will write the predictions to.\n", + "- `parameters`: Additional filtering parameters for serving prediction results.\n", + "\n", + "The helper function calls the job client service's `create_batch_prediction_job` metho, with the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for Dataset, Model and Pipeline resources.\n", + "- `batch_prediction_job`: The specification for the batch prediction job.\n", + "\n", + "Let's now dive into the specification for the `batch_prediction_job`:\n", + "\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `dedicated_resources`: The compute resources to provision for the batch prediction job.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `starting_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "- `model_parameters`: Additional filtering parameters for serving prediction results. *Note*, image segmentation models do not support additional parameters.\n", + "- `input_config`: The input source and format type for the instances to predict.\n", + " - `instances_format`: The format of the batch prediction request file: `jsonl` only supported.\n", + " - `gcs_source`: A list of one or more Cloud Storage paths to your batch prediction requests.\n", + "- `output_config`: The output destination and format for the predictions.\n", + " - `prediction_format`: The format of the batch prediction response file: `jsonl` only supported.\n", + " - `gcs_destination`: The output destination for the predictions.\n", + "\n", + "This call is an asychronous operation. You will print from the response object a few select fields, including:\n", + "\n", + "- `name`: The Vertex fully qualified identifier assigned to the batch prediction job.\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the Model resource.\n", + "- `generate_explanations`: Whether True/False explanations were provided with the predictions (explainability).\n", + "- `state`: The state of the prediction job (pending, running, etc).\n", + "\n", + "Since this call will take a few moments to execute, you will likely get `JobState.JOB_STATE_PENDING` for `state`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_request:automl,isg" + }, + "outputs": [], + "source": [ + "BATCH_MODEL = \"unknown_batch-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_batch_prediction_job(\n", + " display_name,\n", + " model_name,\n", + " gcs_source_uri,\n", + " gcs_destination_output_uri_prefix,\n", + " parameters=None,\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " batch_prediction_job = {\n", + " \"display_name\": display_name,\n", + " # Format: 'projects/{project}/locations/{location}/models/{model_id}'\n", + " \"model\": model_name,\n", + " \"model_parameters\": json_format.ParseDict(parameters, Value()),\n", + " \"input_config\": {\n", + " \"instances_format\": IN_FORMAT,\n", + " \"gcs_source\": {\"uris\": [gcs_source_uri]},\n", + " },\n", + " \"output_config\": {\n", + " \"predictions_format\": OUT_FORMAT,\n", + " \"gcs_destination\": {\"output_uri_prefix\": gcs_destination_output_uri_prefix},\n", + " },\n", + " \"dedicated_resources\": {\n", + " \"machine_spec\": machine_spec,\n", + " \"starting_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " },\n", + " }\n", + " response = clients[\"job\"].create_batch_prediction_job(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + " )\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try:\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", response.labels)\n", + " return response\n", + "\n", + "\n", + "IN_FORMAT = \"jsonl\"\n", + "OUT_FORMAT = \"jsonl\" # [jsonl]\n", + "\n", + "response = create_batch_prediction_job(\n", + " BATCH_MODEL, model_to_deploy_id, gcs_input_uri, BUCKET_NAME, None\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batch_job_id:response" + }, + "source": [ + "Now get the unique identifier for the batch prediction job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batch_job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the batch job\n", + "batch_job_id = response.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split(\"/\")[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_batch_prediction_job" + }, + "source": [ + "### Get information on a batch prediction job\n", + "\n", + "Use this helper function `get_batch_prediction_job`, with the following paramter:\n", + "\n", + "- `job_name`: The Vertex fully qualified identifier for the batch prediction job.\n", + "\n", + "The helper function calls the job client service's `get_batch_prediction_job` method, with the following paramter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the batch prediction job. In this tutorial, you will pass it the Vertex fully qualified identifier for your batch prediction job -- `batch_job_id`\n", + "\n", + "The helper function will return the Cloud Storage path to where the predictions are stored -- `gcs_destination`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_batch_prediction_job" + }, + "outputs": [], + "source": [ + "def get_batch_prediction_job(job_name, silent=False):\n", + " response = clients[\"job\"].get_batch_prediction_job(name=job_name)\n", + " if silent:\n", + " return response.output_config.gcs_destination.output_uri_prefix, response.state\n", + "\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try: # not all data types support explanations\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" error:\", response.error)\n", + " gcs_destination = response.output_config.gcs_destination\n", + " print(\" gcs_destination\")\n", + " print(\" output_uri_prefix:\", gcs_destination.output_uri_prefix)\n", + " return gcs_destination.output_uri_prefix, response.state\n", + "\n", + "\n", + "predictions, state = get_batch_prediction_job(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_the_predictions:automl,isg" + }, + "source": [ + "### Get the predictions\n", + "\n", + "When the batch prediction is done processing, the job state will be `JOB_STATE_SUCCEEDED`.\n", + "\n", + "Finally you view the predictions stored at the Cloud Storage path you set as output. The predictions will be in a JSONL format, which you indicated at the time you made the batch prediction job, under a subfolder starting with the name `prediction`, and under that folder will be a file called `predictions*.jsonl`.\n", + "\n", + "Now display (cat) the contents. You will see multiple JSON objects, one for each prediction.\n", + "\n", + "The first field `ID` is the image file you did the prediction on, and the second field `annotations` is the prediction, which is further broken down into:\n", + "\n", + "- `confidenceMask`: PNG pixel mask indicating confidence in prediction per pixel.\n", + "- `categoryMask`: PNG pixel mask indicating prediction per pixel." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_the_predictions:automl,image" + }, + "outputs": [], + "source": [ + "def get_latest_predictions(gcs_out_dir):\n", + " \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n", + " folders = !gsutil ls $gcs_out_dir\n", + " latest = \"\"\n", + " for folder in folders:\n", + " subfolder = folder.split(\"/\")[-2]\n", + " if subfolder.startswith(\"prediction-\"):\n", + " if subfolder > latest:\n", + " latest = folder[:-1]\n", + " return latest\n", + "\n", + "\n", + "while True:\n", + " predictions, state = get_batch_prediction_job(batch_job_id, True)\n", + " if state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", state)\n", + " if state == aip.JobState.JOB_STATE_FAILED:\n", + " raise Exception(\"Batch Job Failed\")\n", + " else:\n", + " folder = get_latest_predictions(predictions)\n", + " ! gsutil ls $folder/prediction*.jsonl\n", + "\n", + " ! gsutil cat $folder/prediction*.jsonl\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_automl_image_segmentation_batch.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/automl/showcase_automl_image_segmentation_online.ipynb b/notebooks/community/gapic/automl/showcase_automl_image_segmentation_online.ipynb new file mode 100644 index 000000000..dd557c53b --- /dev/null +++ b/notebooks/community/gapic/automl/showcase_automl_image_segmentation_online.ipynb @@ -0,0 +1,1574 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: AutoML image segmentation model for online prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to create image segmentation models and do online prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:isg_unknown,isg" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [TODO](https://). This dataset does not require any feature engineering. The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,online_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML image segmentation model and deploy for online prediction from a Python script using the Vertex client library. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model`." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Set constants unique to AutoML datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the `Dataset` resource service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the `Dataset` resource service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:isg" + }, + "outputs": [], + "source": [ + "# Image Dataset type\n", + "DATA_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/metadata/image_1.0.0.yaml\"\n", + "# Image Labeling type\n", + "LABEL_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/ioformat/image_segmentation_io_format_1.0.0.yaml\"\n", + "# Image Training task\n", + "TRAINING_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_image_segmentation_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML image segmentation model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:automl,online_prediction" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training.\n", + "- Endpoint Service for deployment.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:automl,online_prediction" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "\n", + "def create_dataset(name, schema, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " dataset = aip.Dataset(\n", + " display_name=name, metadata_schema_uri=schema, labels=labels\n", + " )\n", + "\n", + " operation = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"unknown-\" + TIMESTAMP, DATA_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:image,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for images has some requirements for your data:\n", + "\n", + "- Images must be stored in a Cloud Storage bucket.\n", + "- Each image file must be in an image format (PNG, JPEG, BMP, ...).\n", + "- There must be an index file stored in your Cloud Storage bucket that contains the path and label for each image.\n", + "- The index file must be either CSV or JSONL." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:isg,u_dataset,jsonl" + }, + "source": [ + "#### JSONL\n", + "\n", + "For image segmentation, the JSONL index file has the requirements:\n", + "\n", + "- Each data item is a separate JSON object, on a separate line.\n", + "- The key/value pair `image_gcs_uri` is the Cloud Storage path to the image.\n", + "- The key/value pair `category_mask_uri` is the Cloud Storage path to the mask image in PNG format.\n", + "- The key/value pair `'annotation_spec_colors'` is a list mapping mask colors to a label.\n", + " - The key/value pair pair `display_name` is the label for the pixel color mask.\n", + " - The key/value pair pair `color` are the RGB normalized pixel values (between 0 and 1) of the mask for the corresponding label.\n", + "\n", + " { 'image_gcs_uri': image, 'segmentation_annotations': { 'category_mask_uri': mask_image, 'annotation_spec_colors' : [ { 'display_name': label, 'color': {\"red\": value, \"blue\", value, \"green\": value} }, ...] }\n", + "\n", + "*Note*: The dictionary key fields may alternatively be in camelCase. For example, 'image_gcs_uri' can also be 'imageGcsUri'." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,jsonl" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the JSONL index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:isg_unknown,jsonl,isg" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://ucaip-test-us-central1/dataset/isg_data.jsonl\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:jsonl" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Unknown dataset that is stored in a public Cloud Storage bucket, using a JSONL index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of objects in a JSONL index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:jsonl" + }, + "outputs": [], + "source": [ + "if \"IMPORT_FILES\" in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_data" + }, + "source": [ + "### Import data\n", + "\n", + "Now, import the data into your Vertex Dataset resource. Use this helper function `import_data` to import the data. The function does the following:\n", + "\n", + "- Uses the `Dataset` client.\n", + "- Calls the client method `import_data`, with the following parameters:\n", + " - `name`: The human readable name you give to the `Dataset` resource (e.g., unknown).\n", + " - `import_configs`: The import configuration.\n", + "\n", + "- `import_configs`: A Python list containing a dictionary, with the key/value entries:\n", + " - `gcs_sources`: A list of URIs to the paths of the one or more index files.\n", + " - `import_schema_uri`: The schema identifying the labeling type.\n", + "\n", + "The `import_data()` method returns a long running `operation` object. This will take a few minutes to complete. If you are in a live tutorial, this would be a good time to ask questions, or take a personal break." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_data" + }, + "outputs": [], + "source": [ + "def import_data(dataset, gcs_sources, schema):\n", + " config = [{\"gcs_source\": {\"uris\": gcs_sources}, \"import_schema_uri\": schema}]\n", + " print(\"dataset:\", dataset_id)\n", + " start_time = time.time()\n", + " try:\n", + " operation = clients[\"dataset\"].import_data(\n", + " name=dataset_id, import_configs=config\n", + " )\n", + " print(\"Long running operation:\", operation.operation.name)\n", + "\n", + " result = operation.result()\n", + " print(\"result:\", result)\n", + " print(\"time:\", int(time.time() - start_time), \"secs\")\n", + " print(\"error:\", operation.exception())\n", + " print(\"meta :\", operation.metadata)\n", + " print(\n", + " \"after: running:\",\n", + " operation.running(),\n", + " \"done:\",\n", + " operation.done(),\n", + " \"cancelled:\",\n", + " operation.cancelled(),\n", + " )\n", + "\n", + " return operation\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "import_data(dataset_id, [IMPORT_FILE], LABEL_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML image segmentation model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_pipeline:automl" + }, + "source": [ + "### Create a training pipeline\n", + "\n", + "You may ask, what do we use a pipeline for? You typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `pipeline_name`: A human readable name for the pipeline job.\n", + "- `model_name`: A human readable name for the model.\n", + "- `dataset`: The Vertex fully qualified dataset identifier.\n", + "- `schema`: The dataset labeling (annotation) training schema.\n", + "- `task`: A dictionary describing the requirements for the training job.\n", + "\n", + "The helper function calls the `Pipeline` client service'smethod `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "Let's look now deeper into the *minimal* requirements for constructing a `training_pipeline` specification:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The dataset labeling (annotation) training schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A human readable name for the model.\n", + "- `input_data_config`: The dataset specification.\n", + " - `dataset_id`: The Vertex dataset identifier only (non-fully qualified) -- this is the last part of the fully-qualified identifier.\n", + " - `fraction_split`: If specified, the percentages of the dataset to use for training, test and validation. Otherwise, the percentages are automatically selected by AutoML." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_pipeline:automl" + }, + "outputs": [], + "source": [ + "def create_pipeline(pipeline_name, model_name, dataset, schema, task):\n", + "\n", + " dataset_id = dataset.split(\"/\")[-1]\n", + "\n", + " input_config = {\n", + " \"dataset_id\": dataset_id,\n", + " \"fraction_split\": {\n", + " \"training_fraction\": 0.8,\n", + " \"validation_fraction\": 0.1,\n", + " \"test_fraction\": 0.1,\n", + " },\n", + " }\n", + "\n", + " training_pipeline = {\n", + " \"display_name\": pipeline_name,\n", + " \"training_task_definition\": schema,\n", + " \"training_task_inputs\": task,\n", + " \"input_data_config\": input_config,\n", + " \"model_to_upload\": {\"display_name\": model_name},\n", + " }\n", + "\n", + " try:\n", + " pipeline = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " )\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "task_requirements:automl,isg" + }, + "source": [ + "### Construct the task requirements\n", + "\n", + "Next, construct the task requirements. Unlike other parameters which take a Python (JSON-like) dictionary, the `task` field takes a Google protobuf Struct, which is very similar to a Python dictionary. Use the `json_format.ParseDict` method for the conversion.\n", + "\n", + "The minimal fields you need to specify are:\n", + "\n", + "- `budget_milli_node_hours`: The maximum time to budget (billed) for training the model, where 1000 = 1 hour.\n", + "- `model_type`: The type of deployed model:\n", + " - `CLOUD_HIGH_ACCURACY_1`: For deploying to Google Cloud and optimizing for accuracy.\n", + " - `CLOUD_LOW_LATENCY_1`: For deploying to Google Cloud and optimizing for latency (response time),\n", + "\n", + "Finally, create the pipeline by calling the helper function `create_pipeline`, which returns an instance of a training pipeline object." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_requirements:automl,isg" + }, + "outputs": [], + "source": [ + "PIPE_NAME = \"unknown_pipe-\" + TIMESTAMP\n", + "MODEL_NAME = \"unknown_model-\" + TIMESTAMP\n", + "\n", + "task = json_format.ParseDict(\n", + " {\"budget_milli_node_hours\": 2000, \"model_type\": \"CLOUD_LOW_ACCURACY_1\"}, Value()\n", + ")\n", + "\n", + "response = create_pipeline(PIPE_NAME, MODEL_NAME, dataset_id, TRAINING_SCHEMA, task)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 30 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "model_information" + }, + "source": [ + "## Model information\n", + "\n", + "Now that your model is trained, you can get some information on your model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:automl" + }, + "source": [ + "## Evaluate the Model resource\n", + "\n", + "Now find out how good the model service believes your model is. As part of training, some portion of the dataset was set aside as the test (holdout) data, which is used by the pipeline service to evaluate the model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "list_model_evaluations:automl,isg" + }, + "source": [ + "### List evaluations for all slices\n", + "\n", + "Use this helper function `list_model_evaluations`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified model identifier for the `Model` resource.\n", + "\n", + "This helper function uses the model client service's `list_model_evaluations` method, which takes the same parameter. The response object from the call is a list, where each element is an evaluation metric.\n", + "\n", + "For each evaluation -- you probably only have one, we then print all the key names for each metric in the evaluation, and for a small set (`confidenceMetricsEntries`) you will print the result." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "list_model_evaluations:automl,isg" + }, + "outputs": [], + "source": [ + "def list_model_evaluations(name):\n", + " response = clients[\"model\"].list_model_evaluations(parent=name)\n", + " for evaluation in response:\n", + " print(\"model_evaluation\")\n", + " print(\" name:\", evaluation.name)\n", + " print(\" metrics_schema_uri:\", evaluation.metrics_schema_uri)\n", + " metrics = json_format.MessageToDict(evaluation._pb.metrics)\n", + " for metric in metrics.keys():\n", + " print(metric)\n", + " print(\"confidenceMetricsEntries\", metrics[\"confidenceMetricsEntries\"])\n", + "\n", + " return evaluation.name\n", + "\n", + "\n", + "last_evaluation = list_model_evaluations(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:automl" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created with AutoML. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"unknown_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:automatic" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `automatic_resources`: This refers to how many redundant compute instances (replicas). For this example, we set it to one (no replication).\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:automatic" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"unknown_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"automatic_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " },\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_item" + }, + "source": [ + "### Get test item\n", + "\n", + "You will use an arbitrary example out of the dataset as a test item. Don't be concerned that the example was likely used in training the model -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:automl,isg,jsonl" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "test_items = !gsutil cat $IMPORT_FILE | head -n1\n", + "test_data = test_items[0].replace(\"'\", '\"')\n", + "test_data = json.loads(test_data)\n", + "try:\n", + " test_item = test_data[\"image_gcs_uri\"]\n", + " test_label = test_data[\"segmentation_annotation\"][\"annotation_spec_colors\"]\n", + "except:\n", + " test_item = test_data[\"imageGcsUri\"]\n", + " test_label = test_data[\"segmentationAnnotation\"][\"annotationSpecColors\"]\n", + "\n", + "print(test_item, test_label)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "predict_item:automl,isg" + }, + "source": [ + "### Make a prediction\n", + "\n", + "Now you have a test item. Use this helper function `predict_item`, which takes the following parameters:\n", + "\n", + "- `filename`: The Cloud Storage path to the test item.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `parameters_dict`: Additional filtering parameters for serving prediction results.\n", + "\n", + "This function calls the prediction client service's `predict` method with the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `instances`: A list of instances (encoded images) to predict.\n", + "- `parameters`: Additional filtering parameters for serving prediction results. *Note*, image segmentation models do not support additional parameters.\n", + "\n", + "#### Request\n", + "\n", + "Since in this example your test item is in a Cloud Storage bucket, you will open and read the contents of the imageusing `tf.io.gfile.Gfile()`. To pass the test data to the prediction service, we will encode the bytes into base 64 -- This makes binary data safe from modification while it is transferred over the Internet.\n", + "\n", + "The format of each instance is:\n", + "\n", + " { 'content': { 'b64': [base64_encoded_bytes] } }\n", + "\n", + "Since the `predict()` method can take multiple items (instances), you send our single test item as a list of one test item. As a final step, you package the instances list into Google's protobuf format -- which is what we pass to the `predict()` method.\n", + "\n", + "#### Response\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding image in the request. You will see in the output for each prediction -- in our case there is just one:\n", + "\n", + "- `confidenceMask`: Confidence level in the prediction.\n", + "- `categoryMask`: The predicted label per pixel." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "predict_item:automl,isg" + }, + "outputs": [], + "source": [ + "import base64\n", + "\n", + "import tensorflow as tf\n", + "\n", + "\n", + "def predict_item(filename, endpoint, parameters_dict):\n", + "\n", + " parameters = json_format.ParseDict(parameters_dict, Value())\n", + " with tf.io.gfile.GFile(filename, \"rb\") as f:\n", + " content = f.read()\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [{\"content\": base64.b64encode(content).decode(\"utf-8\")}]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", dict(prediction))\n", + "\n", + "\n", + "predict_item(test_item, endpoint_id, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_automl_image_segmentation_online.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/automl/showcase_automl_tabular_binary_classification_batch.ipynb b/notebooks/community/gapic/automl/showcase_automl_tabular_binary_classification_batch.ipynb new file mode 100644 index 000000000..fd1a4b6a1 --- /dev/null +++ b/notebooks/community/gapic/automl/showcase_automl_tabular_binary_classification_batch.ipynb @@ -0,0 +1,1811 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: AutoML tabular binary classification model for batch prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to create tabular binary classification models and do batch prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:bank,lbn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Bank Marketing](gs://cloud-ml-tables-data/bank-marketing.csv). This dataset does not require any feature engineering. The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,batch_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML tabular binary classification model from a Python script, and then do a batch prediction using the Vertex client library. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Make a batch prediction.\n", + "\n", + "There is one key difference between using batch prediction and using online prediction:\n", + "\n", + "* Prediction Service: Does an on-demand prediction for the entire set of instances (i.e., one or more data items) and returns the results in real-time.\n", + "\n", + "* Batch Prediction Service: Does a queued (batch) prediction for the entire set of instances in the background and stores the results in a Cloud Storage bucket when ready." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Set constants unique to AutoML datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the `Dataset` resource service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the `Dataset` resource service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:lbn" + }, + "outputs": [], + "source": [ + "# Tabular Dataset type\n", + "DATA_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/metadata/tables_1.0.0.yaml\"\n", + "# Tabular Labeling type\n", + "LABEL_SCHEMA = (\n", + " \"gs://google-cloud-aiplatform/schema/dataset/ioformat/table_io_format_1.0.0.yaml\"\n", + ")\n", + "# Tabular Training task\n", + "TRAINING_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_tables_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:prediction" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for prediction.\n", + "\n", + "Set the variable `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:automl" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "For AutoML batch prediction, the container image for the serving binary is pre-determined by the Vertex prediction service. More specifically, the service will pick the appropriate container for the model depending on the hardware accelerator you selected." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for prediction.\n", + "\n", + "- Set the variable `DEPLOY_COMPUTE` to configure the compute resources for the VM you will use for prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML tabular binary classification model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:automl,batch_prediction" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training.\n", + "- Job Service for batch prediction and custom training." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:automl,batch_prediction" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"job\"] = create_job_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:tabular" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step is to create a `Dataset` resource instance. This step differs from Vision, Video and Language. For those products, after the `Dataset` resource is created, one then separately imports the data, using the `import_data` method.\n", + "\n", + "For tabular, importing of the data is deferred until the training pipeline starts training the model. What do we do different? Well, first you won't be calling the `import_data` method. Instead, when you create the dataset instance you specify the Cloud Storage location of the CSV file or BigQuery location of the data table, which contains your tabular data as part of the `Dataset` resource's metadata.\n", + "\n", + "#### Cloud Storage\n", + "\n", + "`metadata = {\"input_config\": {\"gcs_source\": {\"uri\": [gcs_uri]}}}`\n", + "\n", + "The format for a Cloud Storage path is:\n", + "\n", + " gs://[bucket_name]/[folder(s)/[file]\n", + "\n", + "#### BigQuery\n", + "\n", + "`metadata = {\"input_config\": {\"bigquery_source\": {\"uri\": [gcs_uri]}}}`\n", + "\n", + "The format for a BigQuery path is:\n", + "\n", + " bq://[collection].[dataset].[table]\n", + "\n", + "Note that the `uri` field is a list, whereby you can input multiple CSV files or BigQuery tables when your data is split across files." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:tabular,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for tabular has a couple of requirements for your tabular data.\n", + "\n", + "- Must be in a CSV file or a BigQuery query." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:lbn,u_dataset,csv" + }, + "source": [ + "#### CSV\n", + "\n", + "For tabular binary classification, the CSV file has a few requirements:\n", + "\n", + "- The first row must be the heading -- note how this is different from Vision, Video and Language where the requirement is no heading.\n", + "- All but one column are features.\n", + "- One column is the label, which you will specify when you subsequently create the training pipeline." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:bank,csv,lbn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://cloud-ml-tables-data/bank-marketing.csv\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:tabular" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Bank Marketing dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows.\n", + "\n", + "You also need for training to know the heading name of the label column, which is save as `label_column`. For this dataset, it is the last column in the CSV file." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:tabular" + }, + "outputs": [], + "source": [ + "count = ! gsutil cat $IMPORT_FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $IMPORT_FILE | head\n", + "\n", + "heading = ! gsutil cat $IMPORT_FILE | head -n1\n", + "label_column = str(heading).split(\",\")[-1].split(\"'\")[0]\n", + "print(\"Label Column Name\", label_column)\n", + "if label_column is None:\n", + " raise Exception(\"label column missing\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset:tabular" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + " - `metadata`: The Cloud Storage or BigQuery location of the tabular data.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset:tabular" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "\n", + "def create_dataset(name, schema, src_uri=None, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " if src_uri.startswith(\"gs://\"):\n", + " metadata = {\"input_config\": {\"gcs_source\": {\"uri\": [src_uri]}}}\n", + " elif src_uri.startswith(\"bq://\"):\n", + " metadata = {\"input_config\": {\"bigquery_source\": {\"uri\": [src_uri]}}}\n", + " dataset = aip.Dataset(\n", + " display_name=name,\n", + " metadata_schema_uri=schema,\n", + " labels=labels,\n", + " metadata=json_format.ParseDict(metadata, Value()),\n", + " )\n", + "\n", + " operation = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"bank-\" + TIMESTAMP, DATA_SCHEMA, src_uri=IMPORT_FILE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML tabular binary classification model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_pipeline:automl" + }, + "source": [ + "### Create a training pipeline\n", + "\n", + "You may ask, what do we use a pipeline for? You typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `pipeline_name`: A human readable name for the pipeline job.\n", + "- `model_name`: A human readable name for the model.\n", + "- `dataset`: The Vertex fully qualified dataset identifier.\n", + "- `schema`: The dataset labeling (annotation) training schema.\n", + "- `task`: A dictionary describing the requirements for the training job.\n", + "\n", + "The helper function calls the `Pipeline` client service'smethod `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "Let's look now deeper into the *minimal* requirements for constructing a `training_pipeline` specification:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The dataset labeling (annotation) training schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A human readable name for the model.\n", + "- `input_data_config`: The dataset specification.\n", + " - `dataset_id`: The Vertex dataset identifier only (non-fully qualified) -- this is the last part of the fully-qualified identifier.\n", + " - `fraction_split`: If specified, the percentages of the dataset to use for training, test and validation. Otherwise, the percentages are automatically selected by AutoML." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_pipeline:automl" + }, + "outputs": [], + "source": [ + "def create_pipeline(pipeline_name, model_name, dataset, schema, task):\n", + "\n", + " dataset_id = dataset.split(\"/\")[-1]\n", + "\n", + " input_config = {\n", + " \"dataset_id\": dataset_id,\n", + " \"fraction_split\": {\n", + " \"training_fraction\": 0.8,\n", + " \"validation_fraction\": 0.1,\n", + " \"test_fraction\": 0.1,\n", + " },\n", + " }\n", + "\n", + " training_pipeline = {\n", + " \"display_name\": pipeline_name,\n", + " \"training_task_definition\": schema,\n", + " \"training_task_inputs\": task,\n", + " \"input_data_config\": input_config,\n", + " \"model_to_upload\": {\"display_name\": model_name},\n", + " }\n", + "\n", + " try:\n", + " pipeline = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " )\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "task_requirements:automl,tabular" + }, + "source": [ + "### Construct the task requirements\n", + "\n", + "Next, construct the task requirements. Unlike other parameters which take a Python (JSON-like) dictionary, the `task` field takes a Google protobuf Struct, which is very similar to a Python dictionary. Use the `json_format.ParseDict` method for the conversion.\n", + "\n", + "The minimal fields you need to specify are:\n", + "\n", + "- `prediction_type`: Whether we are doing \"classification\" or \"regression\".\n", + "- `target_column`: The CSV heading column name for the column we want to predict (i.e., the label).\n", + "- `train_budget_milli_node_hours`: The maximum time to budget (billed) for training the model, where 1000 = 1 hour.\n", + "- `disable_early_stopping`: Whether True/False to let AutoML use its judgement to stop training early or train for the entire budget.\n", + "- `transformations`: Specifies the feature engineering for each feature column.\n", + "\n", + "For `transformations`, the list must have an entry for each column. The outer key field indicates the type of feature engineering for the corresponding column. In this tutorial, you set it to `\"auto\"` to tell AutoML to automatically determine it.\n", + "\n", + "Finally, create the pipeline by calling the helper function `create_pipeline`, which returns an instance of a training pipeline object." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_transformations:automl,tabular,bank" + }, + "outputs": [], + "source": [ + "TRANSFORMATIONS = [\n", + " {\"auto\": {\"column_name\": \"Age\"}},\n", + " {\"auto\": {\"column_name\": \"Job\"}},\n", + " {\"auto\": {\"column_name\": \"MaritalStatus\"}},\n", + " {\"auto\": {\"column_name\": \"Education\"}},\n", + " {\"auto\": {\"column_name\": \"Default\"}},\n", + " {\"auto\": {\"column_name\": \"Balance\"}},\n", + " {\"auto\": {\"column_name\": \"Housing\"}},\n", + " {\"auto\": {\"column_name\": \"Loan\"}},\n", + " {\"auto\": {\"column_name\": \"Contact\"}},\n", + " {\"auto\": {\"column_name\": \"Day\"}},\n", + " {\"auto\": {\"column_name\": \"Month\"}},\n", + " {\"auto\": {\"column_name\": \"Duration\"}},\n", + " {\"auto\": {\"column_name\": \"Campaign\"}},\n", + " {\"auto\": {\"column_name\": \"PDays\"}},\n", + " {\"auto\": {\"column_name\": \"POutcome\"}},\n", + "]" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_requirements:automl,tabular,transformations" + }, + "outputs": [], + "source": [ + "PIPE_NAME = \"bank_pipe-\" + TIMESTAMP\n", + "MODEL_NAME = \"bank_model-\" + TIMESTAMP\n", + "\n", + "task = Value(\n", + " struct_value=Struct(\n", + " fields={\n", + " \"target_column\": Value(string_value=label_column),\n", + " \"prediction_type\": Value(string_value=\"classification\"),\n", + " \"train_budget_milli_node_hours\": Value(number_value=1000),\n", + " \"disable_early_stopping\": Value(bool_value=False),\n", + " \"transformations\": json_format.ParseDict(TRANSFORMATIONS, Value()),\n", + " }\n", + " )\n", + ")\n", + "\n", + "response = create_pipeline(PIPE_NAME, MODEL_NAME, dataset_id, TRAINING_SCHEMA, task)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 30 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "model_information" + }, + "source": [ + "## Model information\n", + "\n", + "Now that your model is trained, you can get some information on your model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:automl" + }, + "source": [ + "## Evaluate the Model resource\n", + "\n", + "Now find out how good the model service believes your model is. As part of training, some portion of the dataset was set aside as the test (holdout) data, which is used by the pipeline service to evaluate the model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "list_model_evaluations:automl,lbn" + }, + "source": [ + "### List evaluations for all slices\n", + "\n", + "Use this helper function `list_model_evaluations`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified model identifier for the `Model` resource.\n", + "\n", + "This helper function uses the model client service's `list_model_evaluations` method, which takes the same parameter. The response object from the call is a list, where each element is an evaluation metric.\n", + "\n", + "For each evaluation (you probably only have one) we then print all the key names for each metric in the evaluation, and for a small set (`logLoss` and `auPrc`) you will print the result." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "list_model_evaluations:automl,lbn" + }, + "outputs": [], + "source": [ + "def list_model_evaluations(name):\n", + " response = clients[\"model\"].list_model_evaluations(parent=name)\n", + " for evaluation in response:\n", + " print(\"model_evaluation\")\n", + " print(\" name:\", evaluation.name)\n", + " print(\" metrics_schema_uri:\", evaluation.metrics_schema_uri)\n", + " metrics = json_format.MessageToDict(evaluation._pb.metrics)\n", + " for metric in metrics.keys():\n", + " print(metric)\n", + " print(\"logloss\", metrics[\"logLoss\"])\n", + " print(\"auPrc\", metrics[\"auPrc\"])\n", + "\n", + " return evaluation.name\n", + "\n", + "\n", + "last_evaluation = list_model_evaluations(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy:batch_prediction" + }, + "source": [ + "## Model deployment for batch prediction\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created for batch prediction. This differs from deploying a `Model` resource for on-demand prediction.\n", + "\n", + "For online prediction, you:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource.\n", + "\n", + "3. Make online prediction requests to the `Endpoint` resource.\n", + "\n", + "For batch-prediction, you:\n", + "\n", + "1. Create a batch prediction job.\n", + "\n", + "2. The job service will provision resources for the batch prediction request.\n", + "\n", + "3. The results of the batch prediction request are returned to the caller.\n", + "\n", + "4. The job service will unprovision the resoures for the batch prediction request." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a batch prediction request\n", + "\n", + "Now do a batch prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_test_items:automl,batch_prediction" + }, + "source": [ + "### Make test items\n", + "\n", + "You will use synthetic data as a test data items. Don't be concerned that we are using synthetic data -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_test_items:automl,tabular,bank" + }, + "outputs": [], + "source": [ + "HEADING = \"Age,Job,MaritalStatus,Education,Default,Balance,Housing,Loan,Contact,Day,Month,Duration,Campaign,PDays,Previous,POutcome,Deposit\"\n", + "INSTANCE_1 = {\n", + " \"Age\": \"58\",\n", + " \"Job\": \"managment\",\n", + " \"MaritalStatus\": \"married\",\n", + " \"Education\": \"teritary\",\n", + " \"Default\": \"no\",\n", + " \"Balance\": \"2143\",\n", + " \"Housing\": \"yes\",\n", + " \"Loan\": \"no\",\n", + " \"Contact\": \"unknown\",\n", + " \"Day\": \"5\",\n", + " \"Month\": \"may\",\n", + " \"Duration\": \"261\",\n", + " \"Campaign\": \"1\",\n", + " \"PDays\": \"-1\",\n", + " \"Previous\": 0,\n", + " \"POutcome\": \"unknown\",\n", + "}\n", + "INSTANCE_2 = {\n", + " \"Age\": \"44\",\n", + " \"Job\": \"technician\",\n", + " \"MaritalStatus\": \"single\",\n", + " \"Education\": \"secondary\",\n", + " \"Default\": \"no\",\n", + " \"Balance\": \"39\",\n", + " \"Housing\": \"yes\",\n", + " \"Loan\": \"no\",\n", + " \"Contact\": \"unknown\",\n", + " \"Day\": \"5\",\n", + " \"Month\": \"may\",\n", + " \"Duration\": \"151\",\n", + " \"Campaign\": \"1\",\n", + " \"PDays\": \"-1\",\n", + " \"Previous\": 0,\n", + " \"POutcome\": \"unknown\",\n", + "}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:automl,tabular" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "Now make a batch input file, which you will store in your local Cloud Storage bucket. Unlike image, video and text, the batch input file for tabular is only supported for CSV. For CSV file, you make:\n", + "\n", + "- The first line is the heading with the feature (fields) heading names.\n", + "- Each remaining line is a separate prediction request with the corresponding feature values.\n", + "\n", + "For example:\n", + "\n", + " \"feature_1\", \"feature_2\". ...\n", + " value_1, value_2, ..." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_file:automl,tabular" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "gcs_input_uri = BUCKET_NAME + \"/test.csv\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " f.write(HEADING + \"\\n\")\n", + " f.write(str(INSTANCE_1) + \"\\n\")\n", + " f.write(str(INSTANCE_2) + \"\\n\")\n", + "\n", + "print(gcs_input_uri)\n", + "! gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your batch prediction requests:\n", + "\n", + "- Single Instance: The batch prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The batch prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and batch prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The batch prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_request:automl,lbn" + }, + "source": [ + "### Make batch prediction request\n", + "\n", + "Now that your batch of two test items is ready, let's do the batch request. Use this helper function `create_batch_prediction_job`, with the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the prediction job.\n", + "- `model_name`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `gcs_source_uri`: The Cloud Storage path to the input file -- which you created above.\n", + "- `gcs_destination_output_uri_prefix`: The Cloud Storage path that the service will write the predictions to.\n", + "- `parameters`: Additional filtering parameters for serving prediction results.\n", + "\n", + "The helper function calls the job client service's `create_batch_prediction_job` metho, with the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for Dataset, Model and Pipeline resources.\n", + "- `batch_prediction_job`: The specification for the batch prediction job.\n", + "\n", + "Let's now dive into the specification for the `batch_prediction_job`:\n", + "\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `dedicated_resources`: The compute resources to provision for the batch prediction job.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `starting_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "- `model_parameters`: Additional filtering parameters for serving prediction results. *Note*, image segmentation models do not support additional parameters.\n", + "- `input_config`: The input source and format type for the instances to predict.\n", + " - `instances_format`: The format of the batch prediction request file: `csv` only supported.\n", + " - `gcs_source`: A list of one or more Cloud Storage paths to your batch prediction requests.\n", + "- `output_config`: The output destination and format for the predictions.\n", + " - `prediction_format`: The format of the batch prediction response file: `csv` only supported.\n", + " - `gcs_destination`: The output destination for the predictions.\n", + "\n", + "This call is an asychronous operation. You will print from the response object a few select fields, including:\n", + "\n", + "- `name`: The Vertex fully qualified identifier assigned to the batch prediction job.\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the Model resource.\n", + "- `generate_explanations`: Whether True/False explanations were provided with the predictions (explainability).\n", + "- `state`: The state of the prediction job (pending, running, etc).\n", + "\n", + "Since this call will take a few moments to execute, you will likely get `JobState.JOB_STATE_PENDING` for `state`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_request:automl,lbn" + }, + "outputs": [], + "source": [ + "BATCH_MODEL = \"bank_batch-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_batch_prediction_job(\n", + " display_name,\n", + " model_name,\n", + " gcs_source_uri,\n", + " gcs_destination_output_uri_prefix,\n", + " parameters=None,\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " batch_prediction_job = {\n", + " \"display_name\": display_name,\n", + " # Format: 'projects/{project}/locations/{location}/models/{model_id}'\n", + " \"model\": model_name,\n", + " \"model_parameters\": json_format.ParseDict(parameters, Value()),\n", + " \"input_config\": {\n", + " \"instances_format\": IN_FORMAT,\n", + " \"gcs_source\": {\"uris\": [gcs_source_uri]},\n", + " },\n", + " \"output_config\": {\n", + " \"predictions_format\": OUT_FORMAT,\n", + " \"gcs_destination\": {\"output_uri_prefix\": gcs_destination_output_uri_prefix},\n", + " },\n", + " \"dedicated_resources\": {\n", + " \"machine_spec\": machine_spec,\n", + " \"starting_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " },\n", + " }\n", + " response = clients[\"job\"].create_batch_prediction_job(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + " )\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try:\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", response.labels)\n", + " return response\n", + "\n", + "\n", + "IN_FORMAT = \"csv\"\n", + "OUT_FORMAT = \"csv\" # [csv]\n", + "\n", + "response = create_batch_prediction_job(\n", + " BATCH_MODEL, model_to_deploy_id, gcs_input_uri, BUCKET_NAME, None\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batch_job_id:response" + }, + "source": [ + "Now get the unique identifier for the batch prediction job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batch_job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the batch job\n", + "batch_job_id = response.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split(\"/\")[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_batch_prediction_job" + }, + "source": [ + "### Get information on a batch prediction job\n", + "\n", + "Use this helper function `get_batch_prediction_job`, with the following paramter:\n", + "\n", + "- `job_name`: The Vertex fully qualified identifier for the batch prediction job.\n", + "\n", + "The helper function calls the job client service's `get_batch_prediction_job` method, with the following paramter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the batch prediction job. In this tutorial, you will pass it the Vertex fully qualified identifier for your batch prediction job -- `batch_job_id`\n", + "\n", + "The helper function will return the Cloud Storage path to where the predictions are stored -- `gcs_destination`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_batch_prediction_job" + }, + "outputs": [], + "source": [ + "def get_batch_prediction_job(job_name, silent=False):\n", + " response = clients[\"job\"].get_batch_prediction_job(name=job_name)\n", + " if silent:\n", + " return response.output_config.gcs_destination.output_uri_prefix, response.state\n", + "\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try: # not all data types support explanations\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" error:\", response.error)\n", + " gcs_destination = response.output_config.gcs_destination\n", + " print(\" gcs_destination\")\n", + " print(\" output_uri_prefix:\", gcs_destination.output_uri_prefix)\n", + " return gcs_destination.output_uri_prefix, response.state\n", + "\n", + "\n", + "predictions, state = get_batch_prediction_job(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_the_predictions:automl,lbn" + }, + "source": [ + "### Get Predictions\n", + "\n", + "When the batch prediction is done processing, the job state will be `JOB_STATE_SUCCEEDED`.\n", + "\n", + "Finally you view the predictions stored at the Cloud Storage path you set as output. The predictions will be in a CSV format, which you indicated at the time you made the batch prediction job, under a subfolder starting with the name `prediction`, and under that folder will be a file called `predictions*.csv`.\n", + "\n", + "Now display (cat) the contents. You will see multiple rows, one for each prediction.\n", + "\n", + "For each prediction:\n", + "\n", + "- The first four fields are the values (features) you did the prediction on.\n", + "- The remaining fields are the confidence values, between 0 and 1, for each prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_the_predictions:automl,tabular" + }, + "outputs": [], + "source": [ + "def get_latest_predictions(gcs_out_dir):\n", + " \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n", + " folders = !gsutil ls $gcs_out_dir\n", + " latest = \"\"\n", + " for folder in folders:\n", + " subfolder = folder.split(\"/\")[-2]\n", + " if subfolder.startswith(\"prediction-\"):\n", + " if subfolder > latest:\n", + " latest = folder[:-1]\n", + " return latest\n", + "\n", + "\n", + "while True:\n", + " predictions, state = get_batch_prediction_job(batch_job_id, True)\n", + " if state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", state)\n", + " if state == aip.JobState.JOB_STATE_FAILED:\n", + " raise Exception(\"Batch Job Failed\")\n", + " else:\n", + " folder = get_latest_predictions(predictions)\n", + " ! gsutil ls $folder/prediction*.csv\n", + "\n", + " ! gsutil cat $folder/prediction*.csv\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_automl_tabular_binary_classification_batch.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/automl/showcase_automl_tabular_binary_classification_online.ipynb b/notebooks/community/gapic/automl/showcase_automl_tabular_binary_classification_online.ipynb new file mode 100644 index 000000000..b32f6b23f --- /dev/null +++ b/notebooks/community/gapic/automl/showcase_automl_tabular_binary_classification_online.ipynb @@ -0,0 +1,1687 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: AutoML tabular binary classification model for online prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to create tabular binary classification models and do online prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:bank,lbn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Bank Marketing](gs://cloud-ml-tables-data/bank-marketing.csv). This dataset does not require any feature engineering. The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,online_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML tabular binary classification model and deploy for online prediction from a Python script using the Vertex client library. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model`." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Set constants unique to AutoML datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the `Dataset` resource service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the `Dataset` resource service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:lbn" + }, + "outputs": [], + "source": [ + "# Tabular Dataset type\n", + "DATA_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/metadata/tables_1.0.0.yaml\"\n", + "# Tabular Labeling type\n", + "LABEL_SCHEMA = (\n", + " \"gs://google-cloud-aiplatform/schema/dataset/ioformat/table_io_format_1.0.0.yaml\"\n", + ")\n", + "# Tabular Training task\n", + "TRAINING_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_tables_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:prediction" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for prediction.\n", + "\n", + "Set the variable `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:automl" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "For AutoML batch prediction, the container image for the serving binary is pre-determined by the Vertex prediction service. More specifically, the service will pick the appropriate container for the model depending on the hardware accelerator you selected." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for prediction.\n", + "\n", + "- Set the variable `DEPLOY_COMPUTE` to configure the compute resources for the VM you will use for prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML tabular binary classification model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:automl,online_prediction" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training.\n", + "- Endpoint Service for deployment.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:automl,online_prediction" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:tabular" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step is to create a `Dataset` resource instance. This step differs from Vision, Video and Language. For those products, after the `Dataset` resource is created, one then separately imports the data, using the `import_data` method.\n", + "\n", + "For tabular, importing of the data is deferred until the training pipeline starts training the model. What do we do different? Well, first you won't be calling the `import_data` method. Instead, when you create the dataset instance you specify the Cloud Storage location of the CSV file or BigQuery location of the data table, which contains your tabular data as part of the `Dataset` resource's metadata.\n", + "\n", + "#### Cloud Storage\n", + "\n", + "`metadata = {\"input_config\": {\"gcs_source\": {\"uri\": [gcs_uri]}}}`\n", + "\n", + "The format for a Cloud Storage path is:\n", + "\n", + " gs://[bucket_name]/[folder(s)/[file]\n", + "\n", + "#### BigQuery\n", + "\n", + "`metadata = {\"input_config\": {\"bigquery_source\": {\"uri\": [gcs_uri]}}}`\n", + "\n", + "The format for a BigQuery path is:\n", + "\n", + " bq://[collection].[dataset].[table]\n", + "\n", + "Note that the `uri` field is a list, whereby you can input multiple CSV files or BigQuery tables when your data is split across files." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:tabular,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for tabular has a couple of requirements for your tabular data.\n", + "\n", + "- Must be in a CSV file or a BigQuery query." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:lbn,u_dataset,csv" + }, + "source": [ + "#### CSV\n", + "\n", + "For tabular binary classification, the CSV file has a few requirements:\n", + "\n", + "- The first row must be the heading -- note how this is different from Vision, Video and Language where the requirement is no heading.\n", + "- All but one column are features.\n", + "- One column is the label, which you will specify when you subsequently create the training pipeline." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:bank,csv,lbn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://cloud-ml-tables-data/bank-marketing.csv\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:tabular" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Bank Marketing dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows.\n", + "\n", + "You also need for training to know the heading name of the label column, which is save as `label_column`. For this dataset, it is the last column in the CSV file." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:tabular" + }, + "outputs": [], + "source": [ + "count = ! gsutil cat $IMPORT_FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $IMPORT_FILE | head\n", + "\n", + "heading = ! gsutil cat $IMPORT_FILE | head -n1\n", + "label_column = str(heading).split(\",\")[-1].split(\"'\")[0]\n", + "print(\"Label Column Name\", label_column)\n", + "if label_column is None:\n", + " raise Exception(\"label column missing\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset:tabular" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + " - `metadata`: The Cloud Storage or BigQuery location of the tabular data.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset:tabular" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "\n", + "def create_dataset(name, schema, src_uri=None, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " if src_uri.startswith(\"gs://\"):\n", + " metadata = {\"input_config\": {\"gcs_source\": {\"uri\": [src_uri]}}}\n", + " elif src_uri.startswith(\"bq://\"):\n", + " metadata = {\"input_config\": {\"bigquery_source\": {\"uri\": [src_uri]}}}\n", + " dataset = aip.Dataset(\n", + " display_name=name,\n", + " metadata_schema_uri=schema,\n", + " labels=labels,\n", + " metadata=json_format.ParseDict(metadata, Value()),\n", + " )\n", + "\n", + " operation = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"bank-\" + TIMESTAMP, DATA_SCHEMA, src_uri=IMPORT_FILE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML tabular binary classification model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_pipeline:automl" + }, + "source": [ + "### Create a training pipeline\n", + "\n", + "You may ask, what do we use a pipeline for? You typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `pipeline_name`: A human readable name for the pipeline job.\n", + "- `model_name`: A human readable name for the model.\n", + "- `dataset`: The Vertex fully qualified dataset identifier.\n", + "- `schema`: The dataset labeling (annotation) training schema.\n", + "- `task`: A dictionary describing the requirements for the training job.\n", + "\n", + "The helper function calls the `Pipeline` client service'smethod `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "Let's look now deeper into the *minimal* requirements for constructing a `training_pipeline` specification:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The dataset labeling (annotation) training schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A human readable name for the model.\n", + "- `input_data_config`: The dataset specification.\n", + " - `dataset_id`: The Vertex dataset identifier only (non-fully qualified) -- this is the last part of the fully-qualified identifier.\n", + " - `fraction_split`: If specified, the percentages of the dataset to use for training, test and validation. Otherwise, the percentages are automatically selected by AutoML." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_pipeline:automl" + }, + "outputs": [], + "source": [ + "def create_pipeline(pipeline_name, model_name, dataset, schema, task):\n", + "\n", + " dataset_id = dataset.split(\"/\")[-1]\n", + "\n", + " input_config = {\n", + " \"dataset_id\": dataset_id,\n", + " \"fraction_split\": {\n", + " \"training_fraction\": 0.8,\n", + " \"validation_fraction\": 0.1,\n", + " \"test_fraction\": 0.1,\n", + " },\n", + " }\n", + "\n", + " training_pipeline = {\n", + " \"display_name\": pipeline_name,\n", + " \"training_task_definition\": schema,\n", + " \"training_task_inputs\": task,\n", + " \"input_data_config\": input_config,\n", + " \"model_to_upload\": {\"display_name\": model_name},\n", + " }\n", + "\n", + " try:\n", + " pipeline = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " )\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "task_requirements:automl,tabular" + }, + "source": [ + "### Construct the task requirements\n", + "\n", + "Next, construct the task requirements. Unlike other parameters which take a Python (JSON-like) dictionary, the `task` field takes a Google protobuf Struct, which is very similar to a Python dictionary. Use the `json_format.ParseDict` method for the conversion.\n", + "\n", + "The minimal fields you need to specify are:\n", + "\n", + "- `prediction_type`: Whether we are doing \"classification\" or \"regression\".\n", + "- `target_column`: The CSV heading column name for the column we want to predict (i.e., the label).\n", + "- `train_budget_milli_node_hours`: The maximum time to budget (billed) for training the model, where 1000 = 1 hour.\n", + "- `disable_early_stopping`: Whether True/False to let AutoML use its judgement to stop training early or train for the entire budget.\n", + "- `transformations`: Specifies the feature engineering for each feature column.\n", + "\n", + "For `transformations`, the list must have an entry for each column. The outer key field indicates the type of feature engineering for the corresponding column. In this tutorial, you set it to `\"auto\"` to tell AutoML to automatically determine it.\n", + "\n", + "Finally, create the pipeline by calling the helper function `create_pipeline`, which returns an instance of a training pipeline object." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_transformations:automl,tabular,bank" + }, + "outputs": [], + "source": [ + "TRANSFORMATIONS = [\n", + " {\"auto\": {\"column_name\": \"Age\"}},\n", + " {\"auto\": {\"column_name\": \"Job\"}},\n", + " {\"auto\": {\"column_name\": \"MaritalStatus\"}},\n", + " {\"auto\": {\"column_name\": \"Education\"}},\n", + " {\"auto\": {\"column_name\": \"Default\"}},\n", + " {\"auto\": {\"column_name\": \"Balance\"}},\n", + " {\"auto\": {\"column_name\": \"Housing\"}},\n", + " {\"auto\": {\"column_name\": \"Loan\"}},\n", + " {\"auto\": {\"column_name\": \"Contact\"}},\n", + " {\"auto\": {\"column_name\": \"Day\"}},\n", + " {\"auto\": {\"column_name\": \"Month\"}},\n", + " {\"auto\": {\"column_name\": \"Duration\"}},\n", + " {\"auto\": {\"column_name\": \"Campaign\"}},\n", + " {\"auto\": {\"column_name\": \"PDays\"}},\n", + " {\"auto\": {\"column_name\": \"POutcome\"}},\n", + "]" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_requirements:automl,tabular,transformations" + }, + "outputs": [], + "source": [ + "PIPE_NAME = \"bank_pipe-\" + TIMESTAMP\n", + "MODEL_NAME = \"bank_model-\" + TIMESTAMP\n", + "\n", + "task = Value(\n", + " struct_value=Struct(\n", + " fields={\n", + " \"target_column\": Value(string_value=label_column),\n", + " \"prediction_type\": Value(string_value=\"classification\"),\n", + " \"train_budget_milli_node_hours\": Value(number_value=1000),\n", + " \"disable_early_stopping\": Value(bool_value=False),\n", + " \"transformations\": json_format.ParseDict(TRANSFORMATIONS, Value()),\n", + " }\n", + " )\n", + ")\n", + "\n", + "response = create_pipeline(PIPE_NAME, MODEL_NAME, dataset_id, TRAINING_SCHEMA, task)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 30 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "model_information" + }, + "source": [ + "## Model information\n", + "\n", + "Now that your model is trained, you can get some information on your model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:automl" + }, + "source": [ + "## Evaluate the Model resource\n", + "\n", + "Now find out how good the model service believes your model is. As part of training, some portion of the dataset was set aside as the test (holdout) data, which is used by the pipeline service to evaluate the model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "list_model_evaluations:automl,lbn" + }, + "source": [ + "### List evaluations for all slices\n", + "\n", + "Use this helper function `list_model_evaluations`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified model identifier for the `Model` resource.\n", + "\n", + "This helper function uses the model client service's `list_model_evaluations` method, which takes the same parameter. The response object from the call is a list, where each element is an evaluation metric.\n", + "\n", + "For each evaluation (you probably only have one) we then print all the key names for each metric in the evaluation, and for a small set (`logLoss` and `auPrc`) you will print the result." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "list_model_evaluations:automl,lbn" + }, + "outputs": [], + "source": [ + "def list_model_evaluations(name):\n", + " response = clients[\"model\"].list_model_evaluations(parent=name)\n", + " for evaluation in response:\n", + " print(\"model_evaluation\")\n", + " print(\" name:\", evaluation.name)\n", + " print(\" metrics_schema_uri:\", evaluation.metrics_schema_uri)\n", + " metrics = json_format.MessageToDict(evaluation._pb.metrics)\n", + " for metric in metrics.keys():\n", + " print(metric)\n", + " print(\"logloss\", metrics[\"logLoss\"])\n", + " print(\"auPrc\", metrics[\"auPrc\"])\n", + "\n", + " return evaluation.name\n", + "\n", + "\n", + "last_evaluation = list_model_evaluations(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:automl" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created with AutoML. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"bank_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:dedicated" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `dedicated_resources`: This refers to how many compute instances (replicas) that are scaled for serving prediction requests.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `min_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:dedicated" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"bank_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " \"machine_spec\": machine_spec,\n", + " },\n", + " \"disable_container_logging\": False,\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_test_item:automl,online_prediction" + }, + "source": [ + "### Make test item\n", + "\n", + "You will use synthetic data as a test data item. Don't be concerned that we are using synthetic data -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_test_item:automl,tabular,bank" + }, + "outputs": [], + "source": [ + "INSTANCE = {\n", + " \"Age\": \"58\",\n", + " \"Job\": \"managment\",\n", + " \"MaritalStatus\": \"married\",\n", + " \"Education\": \"teritary\",\n", + " \"Default\": \"no\",\n", + " \"Balance\": \"2143\",\n", + " \"Housing\": \"yes\",\n", + " \"Loan\": \"no\",\n", + " \"Contact\": \"unknown\",\n", + " \"Day\": \"5\",\n", + " \"Month\": \"may\",\n", + " \"Duration\": \"261\",\n", + " \"Campaign\": \"1\",\n", + " \"PDays\": \"-1\",\n", + " \"Previous\": 0,\n", + " \"POutcome\": \"unknown\",\n", + "}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "predict_item:automl,lbn" + }, + "source": [ + "### Make a prediction\n", + "\n", + "Now you have a test item. Use this helper function `predict_item`, which takes the following parameters:\n", + "\n", + "- `filename`: The Cloud Storage path to the test item.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `parameters_dict`: Additional filtering parameters for serving prediction results.\n", + "\n", + "This function calls the prediction client service's `predict` method with the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `instances`: A list of instances (data items) to predict.\n", + "- `parameters`: Additional filtering parameters for serving prediction results. *Note*, tabular models do not support additional parameters.\n", + "\n", + "#### Request\n", + "\n", + "The format of each instance is, where values must be specified as a string:\n", + "\n", + " { 'feature_1': 'value_1', 'feature_2': 'value_2', ... }\n", + "\n", + "Since the `predict()` method can take multiple items (instances), you send your single test item as a list of one test item. As a final step, you package the instances list into Google's protobuf format -- which is what we pass to the `predict()` method.\n", + "\n", + "#### Response\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding image in the request. You will see in the output for each prediction -- in this case there is just one:\n", + "\n", + "- `confidences`: Confidence level in the prediction.\n", + "- `displayNames`: The predicted label." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "predict_item:automl,lbn" + }, + "outputs": [], + "source": [ + "def predict_item(data, endpoint, parameters_dict):\n", + " parameters = json_format.ParseDict(parameters_dict, Value())\n", + "\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [data]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", dict(prediction))\n", + "\n", + "\n", + "predict_item(INSTANCE, endpoint_id, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_automl_tabular_binary_classification_online.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/automl/showcase_automl_tabular_classification_batch.ipynb b/notebooks/community/gapic/automl/showcase_automl_tabular_classification_batch.ipynb new file mode 100644 index 000000000..3a26534b7 --- /dev/null +++ b/notebooks/community/gapic/automl/showcase_automl_tabular_classification_batch.ipynb @@ -0,0 +1,1766 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: AutoML tabular classification model for batch prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to create tabular classification models and do batch prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:iris,lcn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Iris dataset](https://www.tensorflow.org/datasets/catalog/iris) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). This dataset does not require any feature engineering. The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts the type of Iris flower species from a class of three species: setosa, virginica, or versicolor." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,batch_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML tabular classification model from a Python script, and then do a batch prediction using the Vertex client library. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Make a batch prediction.\n", + "\n", + "There is one key difference between using batch prediction and using online prediction:\n", + "\n", + "* Prediction Service: Does an on-demand prediction for the entire set of instances (i.e., one or more data items) and returns the results in real-time.\n", + "\n", + "* Batch Prediction Service: Does a queued (batch) prediction for the entire set of instances in the background and stores the results in a Cloud Storage bucket when ready." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Set constants unique to AutoML datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the `Dataset` resource service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the `Dataset` resource service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:lcn" + }, + "outputs": [], + "source": [ + "# Tabular Dataset type\n", + "DATA_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/metadata/tables_1.0.0.yaml\"\n", + "# Tabular Labeling type\n", + "LABEL_SCHEMA = (\n", + " \"gs://google-cloud-aiplatform/schema/dataset/ioformat/table_io_format_1.0.0.yaml\"\n", + ")\n", + "# Tabular Training task\n", + "TRAINING_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_tables_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:prediction" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for prediction.\n", + "\n", + "Set the variable `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:automl" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "For AutoML batch prediction, the container image for the serving binary is pre-determined by the Vertex prediction service. More specifically, the service will pick the appropriate container for the model depending on the hardware accelerator you selected." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for prediction.\n", + "\n", + "- Set the variable `DEPLOY_COMPUTE` to configure the compute resources for the VM you will use for prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML tabular classification model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:automl,batch_prediction" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training.\n", + "- Job Service for batch prediction and custom training." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:automl,batch_prediction" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"job\"] = create_job_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:tabular" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step is to create a `Dataset` resource instance. This step differs from Vision, Video and Language. For those products, after the `Dataset` resource is created, one then separately imports the data, using the `import_data` method.\n", + "\n", + "For tabular, importing of the data is deferred until the training pipeline starts training the model. What do we do different? Well, first you won't be calling the `import_data` method. Instead, when you create the dataset instance you specify the Cloud Storage location of the CSV file or BigQuery location of the data table, which contains your tabular data as part of the `Dataset` resource's metadata.\n", + "\n", + "#### Cloud Storage\n", + "\n", + "`metadata = {\"input_config\": {\"gcs_source\": {\"uri\": [gcs_uri]}}}`\n", + "\n", + "The format for a Cloud Storage path is:\n", + "\n", + " gs://[bucket_name]/[folder(s)/[file]\n", + "\n", + "#### BigQuery\n", + "\n", + "`metadata = {\"input_config\": {\"bigquery_source\": {\"uri\": [gcs_uri]}}}`\n", + "\n", + "The format for a BigQuery path is:\n", + "\n", + " bq://[collection].[dataset].[table]\n", + "\n", + "Note that the `uri` field is a list, whereby you can input multiple CSV files or BigQuery tables when your data is split across files." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:tabular,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for tabular has a couple of requirements for your tabular data.\n", + "\n", + "- Must be in a CSV file or a BigQuery query." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:lcn,u_dataset,csv" + }, + "source": [ + "#### CSV\n", + "\n", + "For tabular classification, the CSV file has a few requirements:\n", + "\n", + "- The first row must be the heading -- note how this is different from Vision, Video and Language where the requirement is no heading.\n", + "- All but one column are features.\n", + "- One column is the label, which you will specify when you subsequently create the training pipeline." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:iris,csv,lcn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://cloud-samples-data/tables/iris_1000.csv\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:tabular" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Iris dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows.\n", + "\n", + "You also need for training to know the heading name of the label column, which is save as `label_column`. For this dataset, it is the last column in the CSV file." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:tabular" + }, + "outputs": [], + "source": [ + "count = ! gsutil cat $IMPORT_FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $IMPORT_FILE | head\n", + "\n", + "heading = ! gsutil cat $IMPORT_FILE | head -n1\n", + "label_column = str(heading).split(\",\")[-1].split(\"'\")[0]\n", + "print(\"Label Column Name\", label_column)\n", + "if label_column is None:\n", + " raise Exception(\"label column missing\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset:tabular" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + " - `metadata`: The Cloud Storage or BigQuery location of the tabular data.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset:tabular" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "\n", + "def create_dataset(name, schema, src_uri=None, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " if src_uri.startswith(\"gs://\"):\n", + " metadata = {\"input_config\": {\"gcs_source\": {\"uri\": [src_uri]}}}\n", + " elif src_uri.startswith(\"bq://\"):\n", + " metadata = {\"input_config\": {\"bigquery_source\": {\"uri\": [src_uri]}}}\n", + " dataset = aip.Dataset(\n", + " display_name=name,\n", + " metadata_schema_uri=schema,\n", + " labels=labels,\n", + " metadata=json_format.ParseDict(metadata, Value()),\n", + " )\n", + "\n", + " operation = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"iris-\" + TIMESTAMP, DATA_SCHEMA, src_uri=IMPORT_FILE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML tabular classification model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_pipeline:automl" + }, + "source": [ + "### Create a training pipeline\n", + "\n", + "You may ask, what do we use a pipeline for? You typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `pipeline_name`: A human readable name for the pipeline job.\n", + "- `model_name`: A human readable name for the model.\n", + "- `dataset`: The Vertex fully qualified dataset identifier.\n", + "- `schema`: The dataset labeling (annotation) training schema.\n", + "- `task`: A dictionary describing the requirements for the training job.\n", + "\n", + "The helper function calls the `Pipeline` client service'smethod `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "Let's look now deeper into the *minimal* requirements for constructing a `training_pipeline` specification:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The dataset labeling (annotation) training schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A human readable name for the model.\n", + "- `input_data_config`: The dataset specification.\n", + " - `dataset_id`: The Vertex dataset identifier only (non-fully qualified) -- this is the last part of the fully-qualified identifier.\n", + " - `fraction_split`: If specified, the percentages of the dataset to use for training, test and validation. Otherwise, the percentages are automatically selected by AutoML." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_pipeline:automl" + }, + "outputs": [], + "source": [ + "def create_pipeline(pipeline_name, model_name, dataset, schema, task):\n", + "\n", + " dataset_id = dataset.split(\"/\")[-1]\n", + "\n", + " input_config = {\n", + " \"dataset_id\": dataset_id,\n", + " \"fraction_split\": {\n", + " \"training_fraction\": 0.8,\n", + " \"validation_fraction\": 0.1,\n", + " \"test_fraction\": 0.1,\n", + " },\n", + " }\n", + "\n", + " training_pipeline = {\n", + " \"display_name\": pipeline_name,\n", + " \"training_task_definition\": schema,\n", + " \"training_task_inputs\": task,\n", + " \"input_data_config\": input_config,\n", + " \"model_to_upload\": {\"display_name\": model_name},\n", + " }\n", + "\n", + " try:\n", + " pipeline = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " )\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "task_requirements:automl,tabular" + }, + "source": [ + "### Construct the task requirements\n", + "\n", + "Next, construct the task requirements. Unlike other parameters which take a Python (JSON-like) dictionary, the `task` field takes a Google protobuf Struct, which is very similar to a Python dictionary. Use the `json_format.ParseDict` method for the conversion.\n", + "\n", + "The minimal fields you need to specify are:\n", + "\n", + "- `prediction_type`: Whether we are doing \"classification\" or \"regression\".\n", + "- `target_column`: The CSV heading column name for the column we want to predict (i.e., the label).\n", + "- `train_budget_milli_node_hours`: The maximum time to budget (billed) for training the model, where 1000 = 1 hour.\n", + "- `disable_early_stopping`: Whether True/False to let AutoML use its judgement to stop training early or train for the entire budget.\n", + "- `transformations`: Specifies the feature engineering for each feature column.\n", + "\n", + "For `transformations`, the list must have an entry for each column. The outer key field indicates the type of feature engineering for the corresponding column. In this tutorial, you set it to `\"auto\"` to tell AutoML to automatically determine it.\n", + "\n", + "Finally, create the pipeline by calling the helper function `create_pipeline`, which returns an instance of a training pipeline object." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_transformations:automl,tabular,iris" + }, + "outputs": [], + "source": [ + "TRANSFORMATIONS = [\n", + " {\"auto\": {\"column_name\": \"sepal_width\"}},\n", + " {\"auto\": {\"column_name\": \"sepal_length\"}},\n", + " {\"auto\": {\"column_name\": \"petal_length\"}},\n", + " {\"auto\": {\"column_name\": \"petal_width\"}},\n", + "]" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_requirements:automl,tabular,transformations" + }, + "outputs": [], + "source": [ + "PIPE_NAME = \"iris_pipe-\" + TIMESTAMP\n", + "MODEL_NAME = \"iris_model-\" + TIMESTAMP\n", + "\n", + "task = Value(\n", + " struct_value=Struct(\n", + " fields={\n", + " \"target_column\": Value(string_value=label_column),\n", + " \"prediction_type\": Value(string_value=\"classification\"),\n", + " \"train_budget_milli_node_hours\": Value(number_value=1000),\n", + " \"disable_early_stopping\": Value(bool_value=False),\n", + " \"transformations\": json_format.ParseDict(TRANSFORMATIONS, Value()),\n", + " }\n", + " )\n", + ")\n", + "\n", + "response = create_pipeline(PIPE_NAME, MODEL_NAME, dataset_id, TRAINING_SCHEMA, task)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 30 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "model_information" + }, + "source": [ + "## Model information\n", + "\n", + "Now that your model is trained, you can get some information on your model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:automl" + }, + "source": [ + "## Evaluate the Model resource\n", + "\n", + "Now find out how good the model service believes your model is. As part of training, some portion of the dataset was set aside as the test (holdout) data, which is used by the pipeline service to evaluate the model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "list_model_evaluations:automl,lcn" + }, + "source": [ + "### List evaluations for all slices\n", + "\n", + "Use this helper function `list_model_evaluations`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified model identifier for the `Model` resource.\n", + "\n", + "This helper function uses the model client service's `list_model_evaluations` method, which takes the same parameter. The response object from the call is a list, where each element is an evaluation metric.\n", + "\n", + "For each evaluation (you probably only have one) we then print all the key names for each metric in the evaluation, and for a small set (`logLoss` and `auPrc`) you will print the result." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "list_model_evaluations:automl,lcn" + }, + "outputs": [], + "source": [ + "def list_model_evaluations(name):\n", + " response = clients[\"model\"].list_model_evaluations(parent=name)\n", + " for evaluation in response:\n", + " print(\"model_evaluation\")\n", + " print(\" name:\", evaluation.name)\n", + " print(\" metrics_schema_uri:\", evaluation.metrics_schema_uri)\n", + " metrics = json_format.MessageToDict(evaluation._pb.metrics)\n", + " for metric in metrics.keys():\n", + " print(metric)\n", + " print(\"logloss\", metrics[\"logLoss\"])\n", + " print(\"auPrc\", metrics[\"auPrc\"])\n", + "\n", + " return evaluation.name\n", + "\n", + "\n", + "last_evaluation = list_model_evaluations(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy:batch_prediction" + }, + "source": [ + "## Model deployment for batch prediction\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created for batch prediction. This differs from deploying a `Model` resource for on-demand prediction.\n", + "\n", + "For online prediction, you:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource.\n", + "\n", + "3. Make online prediction requests to the `Endpoint` resource.\n", + "\n", + "For batch-prediction, you:\n", + "\n", + "1. Create a batch prediction job.\n", + "\n", + "2. The job service will provision resources for the batch prediction request.\n", + "\n", + "3. The results of the batch prediction request are returned to the caller.\n", + "\n", + "4. The job service will unprovision the resoures for the batch prediction request." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a batch prediction request\n", + "\n", + "Now do a batch prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_test_items:automl,batch_prediction" + }, + "source": [ + "### Make test items\n", + "\n", + "You will use synthetic data as a test data items. Don't be concerned that we are using synthetic data -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_test_items:automl,tabular,iris" + }, + "outputs": [], + "source": [ + "HEADING = \"petal_length,petal_width,sepal_length,sepal_width\"\n", + "INSTANCE_1 = \"1.4,1.3,5.1,2.8\"\n", + "INSTANCE_2 = \"1.5,1.2,4.7,2.4\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:automl,tabular" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "Now make a batch input file, which you will store in your local Cloud Storage bucket. Unlike image, video and text, the batch input file for tabular is only supported for CSV. For CSV file, you make:\n", + "\n", + "- The first line is the heading with the feature (fields) heading names.\n", + "- Each remaining line is a separate prediction request with the corresponding feature values.\n", + "\n", + "For example:\n", + "\n", + " \"feature_1\", \"feature_2\". ...\n", + " value_1, value_2, ..." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_file:automl,tabular" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "gcs_input_uri = BUCKET_NAME + \"/test.csv\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " f.write(HEADING + \"\\n\")\n", + " f.write(str(INSTANCE_1) + \"\\n\")\n", + " f.write(str(INSTANCE_2) + \"\\n\")\n", + "\n", + "print(gcs_input_uri)\n", + "! gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your batch prediction requests:\n", + "\n", + "- Single Instance: The batch prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The batch prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and batch prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The batch prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_request:automl,lcn" + }, + "source": [ + "### Make batch prediction request\n", + "\n", + "Now that your batch of two test items is ready, let's do the batch request. Use this helper function `create_batch_prediction_job`, with the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the prediction job.\n", + "- `model_name`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `gcs_source_uri`: The Cloud Storage path to the input file -- which you created above.\n", + "- `gcs_destination_output_uri_prefix`: The Cloud Storage path that the service will write the predictions to.\n", + "- `parameters`: Additional filtering parameters for serving prediction results.\n", + "\n", + "The helper function calls the job client service's `create_batch_prediction_job` metho, with the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for Dataset, Model and Pipeline resources.\n", + "- `batch_prediction_job`: The specification for the batch prediction job.\n", + "\n", + "Let's now dive into the specification for the `batch_prediction_job`:\n", + "\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `dedicated_resources`: The compute resources to provision for the batch prediction job.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `starting_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "- `model_parameters`: Additional filtering parameters for serving prediction results. *Note*, image segmentation models do not support additional parameters.\n", + "- `input_config`: The input source and format type for the instances to predict.\n", + " - `instances_format`: The format of the batch prediction request file: `csv` only supported.\n", + " - `gcs_source`: A list of one or more Cloud Storage paths to your batch prediction requests.\n", + "- `output_config`: The output destination and format for the predictions.\n", + " - `prediction_format`: The format of the batch prediction response file: `csv` only supported.\n", + " - `gcs_destination`: The output destination for the predictions.\n", + "\n", + "This call is an asychronous operation. You will print from the response object a few select fields, including:\n", + "\n", + "- `name`: The Vertex fully qualified identifier assigned to the batch prediction job.\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the Model resource.\n", + "- `generate_explanations`: Whether True/False explanations were provided with the predictions (explainability).\n", + "- `state`: The state of the prediction job (pending, running, etc).\n", + "\n", + "Since this call will take a few moments to execute, you will likely get `JobState.JOB_STATE_PENDING` for `state`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_request:automl,lcn" + }, + "outputs": [], + "source": [ + "BATCH_MODEL = \"iris_batch-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_batch_prediction_job(\n", + " display_name,\n", + " model_name,\n", + " gcs_source_uri,\n", + " gcs_destination_output_uri_prefix,\n", + " parameters=None,\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " batch_prediction_job = {\n", + " \"display_name\": display_name,\n", + " # Format: 'projects/{project}/locations/{location}/models/{model_id}'\n", + " \"model\": model_name,\n", + " \"model_parameters\": json_format.ParseDict(parameters, Value()),\n", + " \"input_config\": {\n", + " \"instances_format\": IN_FORMAT,\n", + " \"gcs_source\": {\"uris\": [gcs_source_uri]},\n", + " },\n", + " \"output_config\": {\n", + " \"predictions_format\": OUT_FORMAT,\n", + " \"gcs_destination\": {\"output_uri_prefix\": gcs_destination_output_uri_prefix},\n", + " },\n", + " \"dedicated_resources\": {\n", + " \"machine_spec\": machine_spec,\n", + " \"starting_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " },\n", + " }\n", + " response = clients[\"job\"].create_batch_prediction_job(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + " )\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try:\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", response.labels)\n", + " return response\n", + "\n", + "\n", + "IN_FORMAT = \"csv\"\n", + "OUT_FORMAT = \"csv\" # [csv]\n", + "\n", + "response = create_batch_prediction_job(\n", + " BATCH_MODEL, model_to_deploy_id, gcs_input_uri, BUCKET_NAME, None\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batch_job_id:response" + }, + "source": [ + "Now get the unique identifier for the batch prediction job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batch_job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the batch job\n", + "batch_job_id = response.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split(\"/\")[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_batch_prediction_job" + }, + "source": [ + "### Get information on a batch prediction job\n", + "\n", + "Use this helper function `get_batch_prediction_job`, with the following paramter:\n", + "\n", + "- `job_name`: The Vertex fully qualified identifier for the batch prediction job.\n", + "\n", + "The helper function calls the job client service's `get_batch_prediction_job` method, with the following paramter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the batch prediction job. In this tutorial, you will pass it the Vertex fully qualified identifier for your batch prediction job -- `batch_job_id`\n", + "\n", + "The helper function will return the Cloud Storage path to where the predictions are stored -- `gcs_destination`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_batch_prediction_job" + }, + "outputs": [], + "source": [ + "def get_batch_prediction_job(job_name, silent=False):\n", + " response = clients[\"job\"].get_batch_prediction_job(name=job_name)\n", + " if silent:\n", + " return response.output_config.gcs_destination.output_uri_prefix, response.state\n", + "\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try: # not all data types support explanations\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" error:\", response.error)\n", + " gcs_destination = response.output_config.gcs_destination\n", + " print(\" gcs_destination\")\n", + " print(\" output_uri_prefix:\", gcs_destination.output_uri_prefix)\n", + " return gcs_destination.output_uri_prefix, response.state\n", + "\n", + "\n", + "predictions, state = get_batch_prediction_job(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_the_predictions:automl,lcn" + }, + "source": [ + "### Get Predictions\n", + "\n", + "When the batch prediction is done processing, the job state will be `JOB_STATE_SUCCEEDED`.\n", + "\n", + "Finally you view the predictions stored at the Cloud Storage path you set as output. The predictions will be in a CSV format, which you indicated at the time you made the batch prediction job, under a subfolder starting with the name `prediction`, and under that folder will be a file called `predictions*.csv`.\n", + "\n", + "Now display (cat) the contents. You will see multiple rows, one for each prediction.\n", + "\n", + "For each prediction:\n", + "\n", + "- The first four fields are the values (features) you did the prediction on.\n", + "- The remaining fields are the confidence values, between 0 and 1, for each prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_the_predictions:automl,tabular" + }, + "outputs": [], + "source": [ + "def get_latest_predictions(gcs_out_dir):\n", + " \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n", + " folders = !gsutil ls $gcs_out_dir\n", + " latest = \"\"\n", + " for folder in folders:\n", + " subfolder = folder.split(\"/\")[-2]\n", + " if subfolder.startswith(\"prediction-\"):\n", + " if subfolder > latest:\n", + " latest = folder[:-1]\n", + " return latest\n", + "\n", + "\n", + "while True:\n", + " predictions, state = get_batch_prediction_job(batch_job_id, True)\n", + " if state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", state)\n", + " if state == aip.JobState.JOB_STATE_FAILED:\n", + " raise Exception(\"Batch Job Failed\")\n", + " else:\n", + " folder = get_latest_predictions(predictions)\n", + " ! gsutil ls $folder/prediction*.csv\n", + "\n", + " ! gsutil cat $folder/prediction*.csv\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_automl_tabular_classification_batch.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/automl/showcase_automl_tabular_classification_batch_explain.ipynb b/notebooks/community/gapic/automl/showcase_automl_tabular_classification_batch_explain.ipynb new file mode 100644 index 000000000..6efcc5334 --- /dev/null +++ b/notebooks/community/gapic/automl/showcase_automl_tabular_classification_batch_explain.ipynb @@ -0,0 +1,1765 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: AutoML tabular classification model for batch prediction with explanation\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl,xai" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to create tabular classification models and do batch prediction with explanation using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:iris,lcn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Iris dataset](https://www.tensorflow.org/datasets/catalog/iris) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). This dataset does not require any feature engineering. The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts the type of Iris flower species from a class of three species: setosa, virginica, or versicolor." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,batch_prediction,xai" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML tabular classification model from a Python script, and then do a batch prediction with explainability using the Vertex client library. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Make a batch prediction with explainability.\n", + "\n", + "There is one key difference between using batch prediction and using online prediction:\n", + "\n", + "* Prediction Service: Does an on-demand prediction for the entire set of instances (i.e., one or more data items) and returns the results in real-time.\n", + "\n", + "* Batch Prediction Service: Does a queued (batch) prediction for the entire set of instances in the background and stores the results in a Cloud Storage bucket when ready." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:v1beta1,protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:v1beta1,protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "import google.cloud.aiplatform_v1beta1 as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Set constants unique to AutoML datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the `Dataset` resource service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the `Dataset` resource service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:lcn" + }, + "outputs": [], + "source": [ + "# Tabular Dataset type\n", + "DATA_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/metadata/tables_1.0.0.yaml\"\n", + "# Tabular Labeling type\n", + "LABEL_SCHEMA = (\n", + " \"gs://google-cloud-aiplatform/schema/dataset/ioformat/table_io_format_1.0.0.yaml\"\n", + ")\n", + "# Tabular Training task\n", + "TRAINING_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_tables_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:prediction" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for prediction.\n", + "\n", + "Set the variable `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:automl" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "For AutoML batch prediction, the container image for the serving binary is pre-determined by the Vertex prediction service. More specifically, the service will pick the appropriate container for the model depending on the hardware accelerator you selected." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for prediction.\n", + "\n", + "- Set the variable `DEPLOY_COMPUTE` to configure the compute resources for the VM you will use for prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML tabular classification model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:automl,batch_prediction" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training.\n", + "- Job Service for batch prediction and custom training." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:automl,batch_prediction" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"job\"] = create_job_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:tabular" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step is to create a `Dataset` resource instance. This step differs from Vision, Video and Language. For those products, after the `Dataset` resource is created, one then separately imports the data, using the `import_data` method.\n", + "\n", + "For tabular, importing of the data is deferred until the training pipeline starts training the model. What do we do different? Well, first you won't be calling the `import_data` method. Instead, when you create the dataset instance you specify the Cloud Storage location of the CSV file or BigQuery location of the data table, which contains your tabular data as part of the `Dataset` resource's metadata.\n", + "\n", + "#### Cloud Storage\n", + "\n", + "`metadata = {\"input_config\": {\"gcs_source\": {\"uri\": [gcs_uri]}}}`\n", + "\n", + "The format for a Cloud Storage path is:\n", + "\n", + " gs://[bucket_name]/[folder(s)/[file]\n", + "\n", + "#### BigQuery\n", + "\n", + "`metadata = {\"input_config\": {\"bigquery_source\": {\"uri\": [gcs_uri]}}}`\n", + "\n", + "The format for a BigQuery path is:\n", + "\n", + " bq://[collection].[dataset].[table]\n", + "\n", + "Note that the `uri` field is a list, whereby you can input multiple CSV files or BigQuery tables when your data is split across files." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:tabular,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for tabular has a couple of requirements for your tabular data.\n", + "\n", + "- Must be in a CSV file or a BigQuery query." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:lcn,u_dataset,csv" + }, + "source": [ + "#### CSV\n", + "\n", + "For tabular classification, the CSV file has a few requirements:\n", + "\n", + "- The first row must be the heading -- note how this is different from Vision, Video and Language where the requirement is no heading.\n", + "- All but one column are features.\n", + "- One column is the label, which you will specify when you subsequently create the training pipeline." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:iris,csv,lcn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://cloud-samples-data/tables/iris_1000.csv\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:tabular" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Iris dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows.\n", + "\n", + "You also need for training to know the heading name of the label column, which is save as `label_column`. For this dataset, it is the last column in the CSV file." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:tabular" + }, + "outputs": [], + "source": [ + "count = ! gsutil cat $IMPORT_FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $IMPORT_FILE | head\n", + "\n", + "heading = ! gsutil cat $IMPORT_FILE | head -n1\n", + "label_column = str(heading).split(\",\")[-1].split(\"'\")[0]\n", + "print(\"Label Column Name\", label_column)\n", + "if label_column is None:\n", + " raise Exception(\"label column missing\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset:tabular" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + " - `metadata`: The Cloud Storage or BigQuery location of the tabular data.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset:tabular" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "\n", + "def create_dataset(name, schema, src_uri=None, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " if src_uri.startswith(\"gs://\"):\n", + " metadata = {\"input_config\": {\"gcs_source\": {\"uri\": [src_uri]}}}\n", + " elif src_uri.startswith(\"bq://\"):\n", + " metadata = {\"input_config\": {\"bigquery_source\": {\"uri\": [src_uri]}}}\n", + " dataset = aip.Dataset(\n", + " display_name=name,\n", + " metadata_schema_uri=schema,\n", + " labels=labels,\n", + " metadata=json_format.ParseDict(metadata, Value()),\n", + " )\n", + "\n", + " operation = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"iris-\" + TIMESTAMP, DATA_SCHEMA, src_uri=IMPORT_FILE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML tabular classification model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_pipeline:automl" + }, + "source": [ + "### Create a training pipeline\n", + "\n", + "You may ask, what do we use a pipeline for? You typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `pipeline_name`: A human readable name for the pipeline job.\n", + "- `model_name`: A human readable name for the model.\n", + "- `dataset`: The Vertex fully qualified dataset identifier.\n", + "- `schema`: The dataset labeling (annotation) training schema.\n", + "- `task`: A dictionary describing the requirements for the training job.\n", + "\n", + "The helper function calls the `Pipeline` client service'smethod `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "Let's look now deeper into the *minimal* requirements for constructing a `training_pipeline` specification:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The dataset labeling (annotation) training schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A human readable name for the model.\n", + "- `input_data_config`: The dataset specification.\n", + " - `dataset_id`: The Vertex dataset identifier only (non-fully qualified) -- this is the last part of the fully-qualified identifier.\n", + " - `fraction_split`: If specified, the percentages of the dataset to use for training, test and validation. Otherwise, the percentages are automatically selected by AutoML." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_pipeline:automl" + }, + "outputs": [], + "source": [ + "def create_pipeline(pipeline_name, model_name, dataset, schema, task):\n", + "\n", + " dataset_id = dataset.split(\"/\")[-1]\n", + "\n", + " input_config = {\n", + " \"dataset_id\": dataset_id,\n", + " \"fraction_split\": {\n", + " \"training_fraction\": 0.8,\n", + " \"validation_fraction\": 0.1,\n", + " \"test_fraction\": 0.1,\n", + " },\n", + " }\n", + "\n", + " training_pipeline = {\n", + " \"display_name\": pipeline_name,\n", + " \"training_task_definition\": schema,\n", + " \"training_task_inputs\": task,\n", + " \"input_data_config\": input_config,\n", + " \"model_to_upload\": {\"display_name\": model_name},\n", + " }\n", + "\n", + " try:\n", + " pipeline = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " )\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "task_requirements:automl,tabular" + }, + "source": [ + "### Construct the task requirements\n", + "\n", + "Next, construct the task requirements. Unlike other parameters which take a Python (JSON-like) dictionary, the `task` field takes a Google protobuf Struct, which is very similar to a Python dictionary. Use the `json_format.ParseDict` method for the conversion.\n", + "\n", + "The minimal fields you need to specify are:\n", + "\n", + "- `prediction_type`: Whether we are doing \"classification\" or \"regression\".\n", + "- `target_column`: The CSV heading column name for the column we want to predict (i.e., the label).\n", + "- `train_budget_milli_node_hours`: The maximum time to budget (billed) for training the model, where 1000 = 1 hour.\n", + "- `disable_early_stopping`: Whether True/False to let AutoML use its judgement to stop training early or train for the entire budget.\n", + "- `transformations`: Specifies the feature engineering for each feature column.\n", + "\n", + "For `transformations`, the list must have an entry for each column. The outer key field indicates the type of feature engineering for the corresponding column. In this tutorial, you set it to `\"auto\"` to tell AutoML to automatically determine it.\n", + "\n", + "Finally, create the pipeline by calling the helper function `create_pipeline`, which returns an instance of a training pipeline object." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_transformations:automl,tabular,iris" + }, + "outputs": [], + "source": [ + "TRANSFORMATIONS = [\n", + " {\"auto\": {\"column_name\": \"sepal_width\"}},\n", + " {\"auto\": {\"column_name\": \"sepal_length\"}},\n", + " {\"auto\": {\"column_name\": \"petal_length\"}},\n", + " {\"auto\": {\"column_name\": \"petal_width\"}},\n", + "]" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_requirements:automl,tabular,transformations" + }, + "outputs": [], + "source": [ + "PIPE_NAME = \"iris_pipe-\" + TIMESTAMP\n", + "MODEL_NAME = \"iris_model-\" + TIMESTAMP\n", + "\n", + "task = Value(\n", + " struct_value=Struct(\n", + " fields={\n", + " \"target_column\": Value(string_value=label_column),\n", + " \"prediction_type\": Value(string_value=\"classification\"),\n", + " \"train_budget_milli_node_hours\": Value(number_value=1000),\n", + " \"disable_early_stopping\": Value(bool_value=False),\n", + " \"transformations\": json_format.ParseDict(TRANSFORMATIONS, Value()),\n", + " }\n", + " )\n", + ")\n", + "\n", + "response = create_pipeline(PIPE_NAME, MODEL_NAME, dataset_id, TRAINING_SCHEMA, task)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 30 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "model_information" + }, + "source": [ + "## Model information\n", + "\n", + "Now that your model is trained, you can get some information on your model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:automl" + }, + "source": [ + "## Evaluate the Model resource\n", + "\n", + "Now find out how good the model service believes your model is. As part of training, some portion of the dataset was set aside as the test (holdout) data, which is used by the pipeline service to evaluate the model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "list_model_evaluations:automl,lcn" + }, + "source": [ + "### List evaluations for all slices\n", + "\n", + "Use this helper function `list_model_evaluations`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified model identifier for the `Model` resource.\n", + "\n", + "This helper function uses the model client service's `list_model_evaluations` method, which takes the same parameter. The response object from the call is a list, where each element is an evaluation metric.\n", + "\n", + "For each evaluation (you probably only have one) we then print all the key names for each metric in the evaluation, and for a small set (`logLoss` and `auPrc`) you will print the result." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "list_model_evaluations:automl,lcn" + }, + "outputs": [], + "source": [ + "def list_model_evaluations(name):\n", + " response = clients[\"model\"].list_model_evaluations(parent=name)\n", + " for evaluation in response:\n", + " print(\"model_evaluation\")\n", + " print(\" name:\", evaluation.name)\n", + " print(\" metrics_schema_uri:\", evaluation.metrics_schema_uri)\n", + " metrics = json_format.MessageToDict(evaluation._pb.metrics)\n", + " for metric in metrics.keys():\n", + " print(metric)\n", + " print(\"logloss\", metrics[\"logLoss\"])\n", + " print(\"auPrc\", metrics[\"auPrc\"])\n", + "\n", + " return evaluation.name\n", + "\n", + "\n", + "last_evaluation = list_model_evaluations(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy:batch_prediction" + }, + "source": [ + "## Model deployment for batch prediction\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created for batch prediction. This differs from deploying a `Model` resource for on-demand prediction.\n", + "\n", + "For online prediction, you:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource.\n", + "\n", + "3. Make online prediction requests to the `Endpoint` resource.\n", + "\n", + "For batch-prediction, you:\n", + "\n", + "1. Create a batch prediction job.\n", + "\n", + "2. The job service will provision resources for the batch prediction request.\n", + "\n", + "3. The results of the batch prediction request are returned to the caller.\n", + "\n", + "4. The job service will unprovision the resoures for the batch prediction request." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a batch prediction request\n", + "\n", + "Now do a batch prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_test_items:automl,batch_prediction" + }, + "source": [ + "### Make test items\n", + "\n", + "You will use synthetic data as a test data items. Don't be concerned that we are using synthetic data -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_test_items:automl,tabular,iris" + }, + "outputs": [], + "source": [ + "HEADING = \"petal_length,petal_width,sepal_length,sepal_width\"\n", + "INSTANCE_1 = \"1.4,1.3,5.1,2.8\"\n", + "INSTANCE_2 = \"1.5,1.2,4.7,2.4\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:automl,tabular" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "Now make a batch input file, which you will store in your local Cloud Storage bucket. Unlike image, video and text, the batch input file for tabular is only supported for CSV. For CSV file, you make:\n", + "\n", + "- The first line is the heading with the feature (fields) heading names.\n", + "- Each remaining line is a separate prediction request with the corresponding feature values.\n", + "\n", + "For example:\n", + "\n", + " \"feature_1\", \"feature_2\". ...\n", + " value_1, value_2, ..." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_file:automl,tabular" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "gcs_input_uri = BUCKET_NAME + \"/test.csv\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " f.write(HEADING + \"\\n\")\n", + " f.write(str(INSTANCE_1) + \"\\n\")\n", + " f.write(str(INSTANCE_2) + \"\\n\")\n", + "\n", + "print(gcs_input_uri)\n", + "! gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your batch prediction requests:\n", + "\n", + "- Single Instance: The batch prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The batch prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and batch prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The batch prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_request:automl,lcn" + }, + "source": [ + "### Make batch prediction request\n", + "\n", + "Now that your batch of two test items is ready, let's do the batch request. Use this helper function `create_batch_prediction_job`, with the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the prediction job.\n", + "- `model_name`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `gcs_source_uri`: The Cloud Storage path to the input file -- which you created above.\n", + "- `gcs_destination_output_uri_prefix`: The Cloud Storage path that the service will write the predictions to.\n", + "- `parameters`: Additional filtering parameters for serving prediction results.\n", + "\n", + "The helper function calls the job client service's `create_batch_prediction_job` metho, with the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for Dataset, Model and Pipeline resources.\n", + "- `batch_prediction_job`: The specification for the batch prediction job.\n", + "\n", + "Let's now dive into the specification for the `batch_prediction_job`:\n", + "\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `dedicated_resources`: The compute resources to provision for the batch prediction job.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `starting_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "- `model_parameters`: Additional filtering parameters for serving prediction results. *Note*, image segmentation models do not support additional parameters.\n", + "- `input_config`: The input source and format type for the instances to predict.\n", + " - `instances_format`: The format of the batch prediction request file: `csv` only supported.\n", + " - `gcs_source`: A list of one or more Cloud Storage paths to your batch prediction requests.\n", + "- `output_config`: The output destination and format for the predictions.\n", + " - `prediction_format`: The format of the batch prediction response file: `csv` only supported.\n", + " - `gcs_destination`: The output destination for the predictions.\n", + "\n", + "This call is an asychronous operation. You will print from the response object a few select fields, including:\n", + "\n", + "- `name`: The Vertex fully qualified identifier assigned to the batch prediction job.\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the Model resource.\n", + "- `generate_explanations`: Whether True/False explanations were provided with the predictions (explainability).\n", + "- `state`: The state of the prediction job (pending, running, etc).\n", + "\n", + "Since this call will take a few moments to execute, you will likely get `JobState.JOB_STATE_PENDING` for `state`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_request:automl,lcn" + }, + "outputs": [], + "source": [ + "BATCH_MODEL = \"iris_batch-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_batch_prediction_job(\n", + " display_name,\n", + " model_name,\n", + " gcs_source_uri,\n", + " gcs_destination_output_uri_prefix,\n", + " parameters=None,\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " batch_prediction_job = {\n", + " \"display_name\": display_name,\n", + " # Format: 'projects/{project}/locations/{location}/models/{model_id}'\n", + " \"model\": model_name,\n", + " \"model_parameters\": json_format.ParseDict(parameters, Value()),\n", + " \"input_config\": {\n", + " \"instances_format\": IN_FORMAT,\n", + " \"gcs_source\": {\"uris\": [gcs_source_uri]},\n", + " },\n", + " \"output_config\": {\n", + " \"predictions_format\": OUT_FORMAT,\n", + " \"gcs_destination\": {\"output_uri_prefix\": gcs_destination_output_uri_prefix},\n", + " },\n", + " \"dedicated_resources\": {\n", + " \"machine_spec\": machine_spec,\n", + " \"starting_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " },\n", + " \"generate_explanation\": True,\n", + " }\n", + " response = clients[\"job\"].create_batch_prediction_job(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + " )\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try:\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", response.labels)\n", + " return response\n", + "\n", + "\n", + "IN_FORMAT = \"csv\"\n", + "OUT_FORMAT = \"csv\" # [csv]\n", + "\n", + "response = create_batch_prediction_job(\n", + " BATCH_MODEL, model_to_deploy_id, gcs_input_uri, BUCKET_NAME, None\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batch_job_id:response" + }, + "source": [ + "Now get the unique identifier for the batch prediction job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batch_job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the batch job\n", + "batch_job_id = response.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split(\"/\")[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_batch_prediction_job" + }, + "source": [ + "### Get information on a batch prediction job\n", + "\n", + "Use this helper function `get_batch_prediction_job`, with the following paramter:\n", + "\n", + "- `job_name`: The Vertex fully qualified identifier for the batch prediction job.\n", + "\n", + "The helper function calls the job client service's `get_batch_prediction_job` method, with the following paramter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the batch prediction job. In this tutorial, you will pass it the Vertex fully qualified identifier for your batch prediction job -- `batch_job_id`\n", + "\n", + "The helper function will return the Cloud Storage path to where the predictions are stored -- `gcs_destination`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_batch_prediction_job" + }, + "outputs": [], + "source": [ + "def get_batch_prediction_job(job_name, silent=False):\n", + " response = clients[\"job\"].get_batch_prediction_job(name=job_name)\n", + " if silent:\n", + " return response.output_config.gcs_destination.output_uri_prefix, response.state\n", + "\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try: # not all data types support explanations\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" error:\", response.error)\n", + " gcs_destination = response.output_config.gcs_destination\n", + " print(\" gcs_destination\")\n", + " print(\" output_uri_prefix:\", gcs_destination.output_uri_prefix)\n", + " return gcs_destination.output_uri_prefix, response.state\n", + "\n", + "\n", + "predictions, state = get_batch_prediction_job(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_the_explanations:automl,tabular" + }, + "source": [ + "### Get the predictions with explanations\n", + "\n", + "When the batch prediction is done processing, the job state will be `JOB_STATE_SUCCEEDED`.\n", + "\n", + "Finally you view the predictions and corresponding explanations stored at the Cloud Storage path you set as output. The explanations will be in a CSV format, which you indicated at the time we made the batch explanation job, under a subfolder starting with the name `prediction`, and under that folder will be a file called `explanations*.csv`.\n", + "\n", + "Now display (cat) the contents. You will see one line for each explanation.\n", + "\n", + "- The first four fields are the values (features) you did the prediction on.\n", + "- The remaining fields are the confidence values, between 0 and 1, for each prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_the_explanations:automl,tabular" + }, + "outputs": [], + "source": [ + "def get_latest_predictions(gcs_out_dir):\n", + " \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n", + " folders = !gsutil ls $gcs_out_dir\n", + " latest = \"\"\n", + " for folder in folders:\n", + " subfolder = folder.split(\"/\")[-2]\n", + " if subfolder.startswith(\"prediction-\"):\n", + " if subfolder > latest:\n", + " latest = folder[:-1]\n", + " return latest\n", + "\n", + "\n", + "while True:\n", + " predictions, state = get_batch_prediction_job(batch_job_id, True)\n", + " if state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", state)\n", + " if state == aip.JobState.JOB_STATE_FAILED:\n", + " raise Exception(\"Batch Job Failed\")\n", + " else:\n", + " folder = get_latest_predictions(predictions)\n", + " ! gsutil ls $folder/explanation*.csv\n", + "\n", + " ! gsutil cat $folder/explanation*.csv\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_automl_tabular_classification_batch_explain.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/automl/showcase_automl_tabular_classification_export_cloud.ipynb b/notebooks/community/gapic/automl/showcase_automl_tabular_classification_export_cloud.ipynb new file mode 100644 index 000000000..498345cbc --- /dev/null +++ b/notebooks/community/gapic/automl/showcase_automl_tabular_classification_export_cloud.ipynb @@ -0,0 +1,1618 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: AutoML tabular classification model for export to cloud\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl,export_cloud" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to create tabular classification models to export as a cloud model using Google Cloud's AutoML." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:iris,lcn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Iris dataset](https://www.tensorflow.org/datasets/catalog/iris) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). This dataset does not require any feature engineering. The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts the type of Iris flower species from a class of three species: setosa, virginica, or versicolor." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,export_cloud" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create a AutoML tabular classification model from a Python script using the Vertex client library, and then export the model in a TensorFlow SavedModel format. You can alternatively create models with AutoML using the the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- Export the model from the `Model` resource to Cloud Storage.\n", + "- Download the model locally.\n", + "- Make a local prediction." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:export_cloud" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for exporting the trained model. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Set constants unique to AutoML datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the `Dataset` resource service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the `Dataset` resource service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:lcn" + }, + "outputs": [], + "source": [ + "# Tabular Dataset type\n", + "DATA_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/metadata/tables_1.0.0.yaml\"\n", + "# Tabular Labeling type\n", + "LABEL_SCHEMA = (\n", + " \"gs://google-cloud-aiplatform/schema/dataset/ioformat/table_io_format_1.0.0.yaml\"\n", + ")\n", + "# Tabular Training task\n", + "TRAINING_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_tables_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_docker" + }, + "source": [ + "### Install Docker (Colab or Local)\n", + "\n", + "If you are using Google Cloud Notebook, Docker is already installed. Skip these steps.\n", + "\n", + "By default, Docker is not installed on Colab. If you're running colab, then you need to do the following to install Docker." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_docker" + }, + "outputs": [], + "source": [ + "if \"google.colab\" in sys.modules:\n", + " ! sudo apt update\n", + " ! sudo apt install apt-transport-https ca-certificates curl software-properties-common\n", + " ! curl -fsSL https://download.docker.com/linux/ubuntu/gpg | sudo apt-key add -\n", + " ! sudo add-apt-repository \"deb [arch=amd64] https://download.docker.com/linux/ubuntu bionic stable\"\n", + " ! sudo apt update\n", + " ! sudo apt install docker-ce" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "start_docker_service" + }, + "source": [ + "#### Start Docker service\n", + "\n", + "Start the `docker` service." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "start_docker_service" + }, + "outputs": [], + "source": [ + "if \"google.colab\" in sys.modules:\n", + " ! sudo service docker start" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML tabular classification model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:automl,export_cloud" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:automl,export_cloud" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:tabular" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step is to create a `Dataset` resource instance. This step differs from Vision, Video and Language. For those products, after the `Dataset` resource is created, one then separately imports the data, using the `import_data` method.\n", + "\n", + "For tabular, importing of the data is deferred until the training pipeline starts training the model. What do we do different? Well, first you won't be calling the `import_data` method. Instead, when you create the dataset instance you specify the Cloud Storage location of the CSV file or BigQuery location of the data table, which contains your tabular data as part of the `Dataset` resource's metadata.\n", + "\n", + "#### Cloud Storage\n", + "\n", + "`metadata = {\"input_config\": {\"gcs_source\": {\"uri\": [gcs_uri]}}}`\n", + "\n", + "The format for a Cloud Storage path is:\n", + "\n", + " gs://[bucket_name]/[folder(s)/[file]\n", + "\n", + "#### BigQuery\n", + "\n", + "`metadata = {\"input_config\": {\"bigquery_source\": {\"uri\": [gcs_uri]}}}`\n", + "\n", + "The format for a BigQuery path is:\n", + "\n", + " bq://[collection].[dataset].[table]\n", + "\n", + "Note that the `uri` field is a list, whereby you can input multiple CSV files or BigQuery tables when your data is split across files." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:tabular,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for tabular has a couple of requirements for your tabular data.\n", + "\n", + "- Must be in a CSV file or a BigQuery query." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:lcn,u_dataset,csv" + }, + "source": [ + "#### CSV\n", + "\n", + "For tabular classification, the CSV file has a few requirements:\n", + "\n", + "- The first row must be the heading -- note how this is different from Vision, Video and Language where the requirement is no heading.\n", + "- All but one column are features.\n", + "- One column is the label, which you will specify when you subsequently create the training pipeline." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:iris,csv,lcn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://cloud-samples-data/tables/iris_1000.csv\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:tabular" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Iris dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows.\n", + "\n", + "You also need for training to know the heading name of the label column, which is save as `label_column`. For this dataset, it is the last column in the CSV file." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:tabular" + }, + "outputs": [], + "source": [ + "count = ! gsutil cat $IMPORT_FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $IMPORT_FILE | head\n", + "\n", + "heading = ! gsutil cat $IMPORT_FILE | head -n1\n", + "label_column = str(heading).split(\",\")[-1].split(\"'\")[0]\n", + "print(\"Label Column Name\", label_column)\n", + "if label_column is None:\n", + " raise Exception(\"label column missing\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset:tabular" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + " - `metadata`: The Cloud Storage or BigQuery location of the tabular data.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset:tabular" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "\n", + "def create_dataset(name, schema, src_uri=None, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " if src_uri.startswith(\"gs://\"):\n", + " metadata = {\"input_config\": {\"gcs_source\": {\"uri\": [src_uri]}}}\n", + " elif src_uri.startswith(\"bq://\"):\n", + " metadata = {\"input_config\": {\"bigquery_source\": {\"uri\": [src_uri]}}}\n", + " dataset = aip.Dataset(\n", + " display_name=name,\n", + " metadata_schema_uri=schema,\n", + " labels=labels,\n", + " metadata=json_format.ParseDict(metadata, Value()),\n", + " )\n", + "\n", + " operation = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"iris-\" + TIMESTAMP, DATA_SCHEMA, src_uri=IMPORT_FILE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML tabular classification model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_pipeline:automl" + }, + "source": [ + "### Create a training pipeline\n", + "\n", + "You may ask, what do we use a pipeline for? You typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `pipeline_name`: A human readable name for the pipeline job.\n", + "- `model_name`: A human readable name for the model.\n", + "- `dataset`: The Vertex fully qualified dataset identifier.\n", + "- `schema`: The dataset labeling (annotation) training schema.\n", + "- `task`: A dictionary describing the requirements for the training job.\n", + "\n", + "The helper function calls the `Pipeline` client service'smethod `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "Let's look now deeper into the *minimal* requirements for constructing a `training_pipeline` specification:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The dataset labeling (annotation) training schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A human readable name for the model.\n", + "- `input_data_config`: The dataset specification.\n", + " - `dataset_id`: The Vertex dataset identifier only (non-fully qualified) -- this is the last part of the fully-qualified identifier.\n", + " - `fraction_split`: If specified, the percentages of the dataset to use for training, test and validation. Otherwise, the percentages are automatically selected by AutoML." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_pipeline:automl" + }, + "outputs": [], + "source": [ + "def create_pipeline(pipeline_name, model_name, dataset, schema, task):\n", + "\n", + " dataset_id = dataset.split(\"/\")[-1]\n", + "\n", + " input_config = {\n", + " \"dataset_id\": dataset_id,\n", + " \"fraction_split\": {\n", + " \"training_fraction\": 0.8,\n", + " \"validation_fraction\": 0.1,\n", + " \"test_fraction\": 0.1,\n", + " },\n", + " }\n", + "\n", + " training_pipeline = {\n", + " \"display_name\": pipeline_name,\n", + " \"training_task_definition\": schema,\n", + " \"training_task_inputs\": task,\n", + " \"input_data_config\": input_config,\n", + " \"model_to_upload\": {\"display_name\": model_name},\n", + " }\n", + "\n", + " try:\n", + " pipeline = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " )\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "task_requirements:automl,tabular" + }, + "source": [ + "### Construct the task requirements\n", + "\n", + "Next, construct the task requirements. Unlike other parameters which take a Python (JSON-like) dictionary, the `task` field takes a Google protobuf Struct, which is very similar to a Python dictionary. Use the `json_format.ParseDict` method for the conversion.\n", + "\n", + "The minimal fields you need to specify are:\n", + "\n", + "- `prediction_type`: Whether we are doing \"classification\" or \"regression\".\n", + "- `target_column`: The CSV heading column name for the column we want to predict (i.e., the label).\n", + "- `train_budget_milli_node_hours`: The maximum time to budget (billed) for training the model, where 1000 = 1 hour.\n", + "- `disable_early_stopping`: Whether True/False to let AutoML use its judgement to stop training early or train for the entire budget.\n", + "- `transformations`: Specifies the feature engineering for each feature column.\n", + "\n", + "For `transformations`, the list must have an entry for each column. The outer key field indicates the type of feature engineering for the corresponding column. In this tutorial, you set it to `\"auto\"` to tell AutoML to automatically determine it.\n", + "\n", + "Finally, create the pipeline by calling the helper function `create_pipeline`, which returns an instance of a training pipeline object." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_transformations:automl,tabular,iris" + }, + "outputs": [], + "source": [ + "TRANSFORMATIONS = [\n", + " {\"auto\": {\"column_name\": \"sepal_width\"}},\n", + " {\"auto\": {\"column_name\": \"sepal_length\"}},\n", + " {\"auto\": {\"column_name\": \"petal_length\"}},\n", + " {\"auto\": {\"column_name\": \"petal_width\"}},\n", + "]" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_requirements:automl,tabular,transformations" + }, + "outputs": [], + "source": [ + "PIPE_NAME = \"iris_pipe-\" + TIMESTAMP\n", + "MODEL_NAME = \"iris_model-\" + TIMESTAMP\n", + "\n", + "task = Value(\n", + " struct_value=Struct(\n", + " fields={\n", + " \"target_column\": Value(string_value=label_column),\n", + " \"prediction_type\": Value(string_value=\"classification\"),\n", + " \"train_budget_milli_node_hours\": Value(number_value=1000),\n", + " \"disable_early_stopping\": Value(bool_value=False),\n", + " \"transformations\": json_format.ParseDict(TRANSFORMATIONS, Value()),\n", + " }\n", + " )\n", + ")\n", + "\n", + "response = create_pipeline(PIPE_NAME, MODEL_NAME, dataset_id, TRAINING_SCHEMA, task)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 30 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "model_information" + }, + "source": [ + "## Model information\n", + "\n", + "Now that your model is trained, you can get some information on your model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:automl" + }, + "source": [ + "## Evaluate the Model resource\n", + "\n", + "Now find out how good the model service believes your model is. As part of training, some portion of the dataset was set aside as the test (holdout) data, which is used by the pipeline service to evaluate the model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "list_model_evaluations:automl,lcn" + }, + "source": [ + "### List evaluations for all slices\n", + "\n", + "Use this helper function `list_model_evaluations`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified model identifier for the `Model` resource.\n", + "\n", + "This helper function uses the model client service's `list_model_evaluations` method, which takes the same parameter. The response object from the call is a list, where each element is an evaluation metric.\n", + "\n", + "For each evaluation (you probably only have one) we then print all the key names for each metric in the evaluation, and for a small set (`logLoss` and `auPrc`) you will print the result." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "list_model_evaluations:automl,lcn" + }, + "outputs": [], + "source": [ + "def list_model_evaluations(name):\n", + " response = clients[\"model\"].list_model_evaluations(parent=name)\n", + " for evaluation in response:\n", + " print(\"model_evaluation\")\n", + " print(\" name:\", evaluation.name)\n", + " print(\" metrics_schema_uri:\", evaluation.metrics_schema_uri)\n", + " metrics = json_format.MessageToDict(evaluation._pb.metrics)\n", + " for metric in metrics.keys():\n", + " print(metric)\n", + " print(\"logloss\", metrics[\"logLoss\"])\n", + " print(\"auPrc\", metrics[\"auPrc\"])\n", + "\n", + " return evaluation.name\n", + "\n", + "\n", + "last_evaluation = list_model_evaluations(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "export_model:export_cloud" + }, + "source": [ + "## Export as cloud model\n", + "\n", + "You can export an AutoML tabular classification model as a TensorFlow SavedFormat model which you can then custom deploy to Google Cloud or download locally. Use this helper function `export_model` to export the model to Google Cloud, which takes the following parameters:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `gcs_dest`: The Cloud Storage location to store the SavedFormat model artifacts to.\n", + "\n", + "This function calls the `Model` client service's method `export_model`, with the following parameters:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `output_config`: The destination information for the exported model.\n", + " - `artifact_destination.output_uri_prefix`: The Cloud Storage location to store the SavedFormat model artifacts to.\n", + " - `export_format_id`: The format to save the model format as. For AutoML tabular classification there is just one option:\n", + " - `tf-saved-model`: TensorFlow SavedFormat\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is exported." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "export_model:export_cloud" + }, + "outputs": [], + "source": [ + "MODEL_DIR = BUCKET_NAME + \"/\" + \"iris\"\n", + "\n", + "\n", + "def export_model(name, format, gcs_dest):\n", + " output_config = {\n", + " \"artifact_destination\": {\"output_uri_prefix\": gcs_dest},\n", + " \"export_format_id\": format,\n", + " }\n", + " response = clients[\"model\"].export_model(name=name, output_config=output_config)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result(timeout=1800)\n", + " metadata = response.operation.metadata\n", + " artifact_uri = str(metadata.value).split(\"\\\\\")[-1][4:-1]\n", + " print(\"Artifact Uri\", artifact_uri)\n", + " return artifact_uri\n", + "\n", + "\n", + "model_package = export_model(model_to_deploy_id, \"tf-saved-model\", MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "export_model_layout:export_cloud" + }, + "source": [ + "### Running the exported model package\n", + "\n", + "Your model is now stored in a TensorFlow SavedModel format, along with a model server binary in a Cloud Storage bucket. Load the model locally from the Cloud Storage bucket, and then you can do some things like deploy the model and serving binary locally using Docker with a prebuilt Google Cloud container designed for running the model server.\n", + "\n", + "#### Model package layout\n", + "\n", + "Let's start with looking at what is in the exported package. The exported model package will be stored in your Cloud Storage bucket as:\n", + "\n", + " //model-/tf-saved-model/\n", + "\n", + "Next, take a look at the contents of the model package." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "export_model_layout:export_cloud" + }, + "outputs": [], + "source": [ + "print(\"Model Package:\", model_package)\n", + "print(\"Contents:\")\n", + "! gsutil ls $model_package\n", + "\n", + "print(\"\\nTF Saved Model\")\n", + "path = model_package + \"/predict\"\n", + "files = ! gsutil ls $path\n", + "saved_dir = files[1]\n", + "print(saved_dir)\n", + "! gsutil ls $saved_dir" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "download_exported_model:export_cloud" + }, + "source": [ + "#### Downloading the export model package.\n", + "\n", + "Your first step is to download the exported model package locally." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "download_exported_model:export_cloud" + }, + "outputs": [], + "source": [ + "! gsutil cp -r $model_package ." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "rename_exported_model:export_cloud" + }, + "source": [ + "#### Rename the model package directory name\n", + "\n", + "In this tutorial, you will launch the model package using Docker. The timestamp in the last component of the path name for the package contains a ':', which is invalid for Docker. You will need to rename. In the next step, we rename the last component to `tbl_exported`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "rename_exported_model:export_cloud" + }, + "outputs": [], + "source": [ + "subdir = model_package.split(\"/\")[-1]\n", + "print(subdir)\n", + "! mv $subdir tbl_exported" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "run_exported_model:local" + }, + "source": [ + "### Run the exported model package\n", + "\n", + "This model package has dependencies on a prebuilt Google Cloud container that is specific for serving an export AutoML tabular classification model:\n", + "\n", + " \"gcr.io/cloud-aiplatform/automl_tables/prediction_server\"\n", + "\n", + "This container is currently not supported by Vertex endpoint. You can though launch the model server with the container, either locally or on another compute instance you manage.\n", + "\n", + "In this tutorial, you will launch it with `docker`. The Docker container will run in the background of this notebook, using detached mode by specifying the `-d` option to `docker`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "run_exported_model:local" + }, + "outputs": [], + "source": [ + "MODEL_SERVER = \"gcr.io/cloud-aiplatform/automl_tables/prediction_server\"\n", + "PORT = 8081\n", + "\n", + "docker_id = ! docker run -d -v `pwd`/tbl_exported:/models/default -p 8081:8080 -it $MODEL_SERVER" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "check_health_status:local" + }, + "source": [ + "#### Check health status\n", + "\n", + "We need to let a bit of time to go by for the model to be loaded. Let's pause 10 seconds and then do a health check. If the health check comes back empty, it is ready to go." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "check_health_status:local" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "time.sleep(10)\n", + "! curl -X GET http://localhost:8081/health" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "docker_verified_loaded:local" + }, + "source": [ + "#### Verify loading of docker image\n", + "\n", + "You can also verify the loading of the Docker image with `docker container ls`. We will add `--latest` to get information on the latest (this image) that was loaded." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "docker_verified_loaded:local" + }, + "outputs": [], + "source": [ + "! docker container ls --latest" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction:automl,export_cloud,tabular" + }, + "source": [ + "## Make a prediction request\n", + "\n", + "Now do a prediction with your exported model.\n", + "\n", + "### Prediction request format\n", + "\n", + "The format for the prediction request is a JSON object of the form:\n", + "\n", + " { \"instances\": [ { \"column_name_1\": value, \"column_name_2\": value, … } , … ] }\n", + "\n", + "Place your prediction request in a text file, such as:\n", + "\n", + " test.json" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_instances_file:tabular,iris" + }, + "outputs": [], + "source": [ + "INSTANCES = {\n", + " \"instances\": [\n", + " {\n", + " \"petal_length\": \"1.4\",\n", + " \"petal_width\": \"1.3\",\n", + " \"sepal_length\": \"5.1\",\n", + " \"sepal_width\": \"2.8\",\n", + " }\n", + " ]\n", + "}\n", + "\n", + "import json\n", + "\n", + "with open(\"test.json\", \"w\") as f:\n", + " data = json.dumps(INSTANCES)\n", + " f.write(data)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "predict_item:local,curl" + }, + "source": [ + "#### Send the prediction request\n", + "\n", + "You can the send the prediction request using CURL:\n", + "\n", + " curl -X POST --data @test.json http://localhost:8081/predict" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "predict_item:local,curl" + }, + "outputs": [], + "source": [ + "! curl -X POST --data @/home/jupyter/test.json http://localhost:8081/predict" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "docker_shutdown:local" + }, + "source": [ + "#### Shutdown the Docker container" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "docker_shutdown:local" + }, + "outputs": [], + "source": [ + "print(docker_id)\n", + "! docker kill $docker_id" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_automl_tabular_classification_export_cloud.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/automl/showcase_automl_tabular_classification_online.ipynb b/notebooks/community/gapic/automl/showcase_automl_tabular_classification_online.ipynb new file mode 100644 index 000000000..2660ee3f6 --- /dev/null +++ b/notebooks/community/gapic/automl/showcase_automl_tabular_classification_online.ipynb @@ -0,0 +1,1664 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: AutoML tabular classification model for online prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to create tabular classification models and do online prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:iris,lcn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Iris dataset](https://www.tensorflow.org/datasets/catalog/iris) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). This dataset does not require any feature engineering. The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts the type of Iris flower species from a class of three species: setosa, virginica, or versicolor." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,online_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML tabular classification model and deploy for online prediction from a Python script using the Vertex client library. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model`." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Set constants unique to AutoML datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the `Dataset` resource service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the `Dataset` resource service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:lcn" + }, + "outputs": [], + "source": [ + "# Tabular Dataset type\n", + "DATA_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/metadata/tables_1.0.0.yaml\"\n", + "# Tabular Labeling type\n", + "LABEL_SCHEMA = (\n", + " \"gs://google-cloud-aiplatform/schema/dataset/ioformat/table_io_format_1.0.0.yaml\"\n", + ")\n", + "# Tabular Training task\n", + "TRAINING_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_tables_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:prediction" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for prediction.\n", + "\n", + "Set the variable `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:automl" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "For AutoML batch prediction, the container image for the serving binary is pre-determined by the Vertex prediction service. More specifically, the service will pick the appropriate container for the model depending on the hardware accelerator you selected." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for prediction.\n", + "\n", + "- Set the variable `DEPLOY_COMPUTE` to configure the compute resources for the VM you will use for prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML tabular classification model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:automl,online_prediction" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training.\n", + "- Endpoint Service for deployment.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:automl,online_prediction" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:tabular" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step is to create a `Dataset` resource instance. This step differs from Vision, Video and Language. For those products, after the `Dataset` resource is created, one then separately imports the data, using the `import_data` method.\n", + "\n", + "For tabular, importing of the data is deferred until the training pipeline starts training the model. What do we do different? Well, first you won't be calling the `import_data` method. Instead, when you create the dataset instance you specify the Cloud Storage location of the CSV file or BigQuery location of the data table, which contains your tabular data as part of the `Dataset` resource's metadata.\n", + "\n", + "#### Cloud Storage\n", + "\n", + "`metadata = {\"input_config\": {\"gcs_source\": {\"uri\": [gcs_uri]}}}`\n", + "\n", + "The format for a Cloud Storage path is:\n", + "\n", + " gs://[bucket_name]/[folder(s)/[file]\n", + "\n", + "#### BigQuery\n", + "\n", + "`metadata = {\"input_config\": {\"bigquery_source\": {\"uri\": [gcs_uri]}}}`\n", + "\n", + "The format for a BigQuery path is:\n", + "\n", + " bq://[collection].[dataset].[table]\n", + "\n", + "Note that the `uri` field is a list, whereby you can input multiple CSV files or BigQuery tables when your data is split across files." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:tabular,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for tabular has a couple of requirements for your tabular data.\n", + "\n", + "- Must be in a CSV file or a BigQuery query." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:lcn,u_dataset,csv" + }, + "source": [ + "#### CSV\n", + "\n", + "For tabular classification, the CSV file has a few requirements:\n", + "\n", + "- The first row must be the heading -- note how this is different from Vision, Video and Language where the requirement is no heading.\n", + "- All but one column are features.\n", + "- One column is the label, which you will specify when you subsequently create the training pipeline." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:iris,csv,lcn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://cloud-samples-data/tables/iris_1000.csv\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:tabular" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Iris dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows.\n", + "\n", + "You also need for training to know the heading name of the label column, which is save as `label_column`. For this dataset, it is the last column in the CSV file." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:tabular" + }, + "outputs": [], + "source": [ + "count = ! gsutil cat $IMPORT_FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $IMPORT_FILE | head\n", + "\n", + "heading = ! gsutil cat $IMPORT_FILE | head -n1\n", + "label_column = str(heading).split(\",\")[-1].split(\"'\")[0]\n", + "print(\"Label Column Name\", label_column)\n", + "if label_column is None:\n", + " raise Exception(\"label column missing\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset:tabular" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + " - `metadata`: The Cloud Storage or BigQuery location of the tabular data.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset:tabular" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "\n", + "def create_dataset(name, schema, src_uri=None, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " if src_uri.startswith(\"gs://\"):\n", + " metadata = {\"input_config\": {\"gcs_source\": {\"uri\": [src_uri]}}}\n", + " elif src_uri.startswith(\"bq://\"):\n", + " metadata = {\"input_config\": {\"bigquery_source\": {\"uri\": [src_uri]}}}\n", + " dataset = aip.Dataset(\n", + " display_name=name,\n", + " metadata_schema_uri=schema,\n", + " labels=labels,\n", + " metadata=json_format.ParseDict(metadata, Value()),\n", + " )\n", + "\n", + " operation = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"iris-\" + TIMESTAMP, DATA_SCHEMA, src_uri=IMPORT_FILE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML tabular classification model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_pipeline:automl" + }, + "source": [ + "### Create a training pipeline\n", + "\n", + "You may ask, what do we use a pipeline for? You typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `pipeline_name`: A human readable name for the pipeline job.\n", + "- `model_name`: A human readable name for the model.\n", + "- `dataset`: The Vertex fully qualified dataset identifier.\n", + "- `schema`: The dataset labeling (annotation) training schema.\n", + "- `task`: A dictionary describing the requirements for the training job.\n", + "\n", + "The helper function calls the `Pipeline` client service'smethod `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "Let's look now deeper into the *minimal* requirements for constructing a `training_pipeline` specification:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The dataset labeling (annotation) training schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A human readable name for the model.\n", + "- `input_data_config`: The dataset specification.\n", + " - `dataset_id`: The Vertex dataset identifier only (non-fully qualified) -- this is the last part of the fully-qualified identifier.\n", + " - `fraction_split`: If specified, the percentages of the dataset to use for training, test and validation. Otherwise, the percentages are automatically selected by AutoML." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_pipeline:automl" + }, + "outputs": [], + "source": [ + "def create_pipeline(pipeline_name, model_name, dataset, schema, task):\n", + "\n", + " dataset_id = dataset.split(\"/\")[-1]\n", + "\n", + " input_config = {\n", + " \"dataset_id\": dataset_id,\n", + " \"fraction_split\": {\n", + " \"training_fraction\": 0.8,\n", + " \"validation_fraction\": 0.1,\n", + " \"test_fraction\": 0.1,\n", + " },\n", + " }\n", + "\n", + " training_pipeline = {\n", + " \"display_name\": pipeline_name,\n", + " \"training_task_definition\": schema,\n", + " \"training_task_inputs\": task,\n", + " \"input_data_config\": input_config,\n", + " \"model_to_upload\": {\"display_name\": model_name},\n", + " }\n", + "\n", + " try:\n", + " pipeline = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " )\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "task_requirements:automl,tabular" + }, + "source": [ + "### Construct the task requirements\n", + "\n", + "Next, construct the task requirements. Unlike other parameters which take a Python (JSON-like) dictionary, the `task` field takes a Google protobuf Struct, which is very similar to a Python dictionary. Use the `json_format.ParseDict` method for the conversion.\n", + "\n", + "The minimal fields you need to specify are:\n", + "\n", + "- `prediction_type`: Whether we are doing \"classification\" or \"regression\".\n", + "- `target_column`: The CSV heading column name for the column we want to predict (i.e., the label).\n", + "- `train_budget_milli_node_hours`: The maximum time to budget (billed) for training the model, where 1000 = 1 hour.\n", + "- `disable_early_stopping`: Whether True/False to let AutoML use its judgement to stop training early or train for the entire budget.\n", + "- `transformations`: Specifies the feature engineering for each feature column.\n", + "\n", + "For `transformations`, the list must have an entry for each column. The outer key field indicates the type of feature engineering for the corresponding column. In this tutorial, you set it to `\"auto\"` to tell AutoML to automatically determine it.\n", + "\n", + "Finally, create the pipeline by calling the helper function `create_pipeline`, which returns an instance of a training pipeline object." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_transformations:automl,tabular,iris" + }, + "outputs": [], + "source": [ + "TRANSFORMATIONS = [\n", + " {\"auto\": {\"column_name\": \"sepal_width\"}},\n", + " {\"auto\": {\"column_name\": \"sepal_length\"}},\n", + " {\"auto\": {\"column_name\": \"petal_length\"}},\n", + " {\"auto\": {\"column_name\": \"petal_width\"}},\n", + "]" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_requirements:automl,tabular,transformations" + }, + "outputs": [], + "source": [ + "PIPE_NAME = \"iris_pipe-\" + TIMESTAMP\n", + "MODEL_NAME = \"iris_model-\" + TIMESTAMP\n", + "\n", + "task = Value(\n", + " struct_value=Struct(\n", + " fields={\n", + " \"target_column\": Value(string_value=label_column),\n", + " \"prediction_type\": Value(string_value=\"classification\"),\n", + " \"train_budget_milli_node_hours\": Value(number_value=1000),\n", + " \"disable_early_stopping\": Value(bool_value=False),\n", + " \"transformations\": json_format.ParseDict(TRANSFORMATIONS, Value()),\n", + " }\n", + " )\n", + ")\n", + "\n", + "response = create_pipeline(PIPE_NAME, MODEL_NAME, dataset_id, TRAINING_SCHEMA, task)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 30 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "model_information" + }, + "source": [ + "## Model information\n", + "\n", + "Now that your model is trained, you can get some information on your model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:automl" + }, + "source": [ + "## Evaluate the Model resource\n", + "\n", + "Now find out how good the model service believes your model is. As part of training, some portion of the dataset was set aside as the test (holdout) data, which is used by the pipeline service to evaluate the model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "list_model_evaluations:automl,lcn" + }, + "source": [ + "### List evaluations for all slices\n", + "\n", + "Use this helper function `list_model_evaluations`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified model identifier for the `Model` resource.\n", + "\n", + "This helper function uses the model client service's `list_model_evaluations` method, which takes the same parameter. The response object from the call is a list, where each element is an evaluation metric.\n", + "\n", + "For each evaluation (you probably only have one) we then print all the key names for each metric in the evaluation, and for a small set (`logLoss` and `auPrc`) you will print the result." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "list_model_evaluations:automl,lcn" + }, + "outputs": [], + "source": [ + "def list_model_evaluations(name):\n", + " response = clients[\"model\"].list_model_evaluations(parent=name)\n", + " for evaluation in response:\n", + " print(\"model_evaluation\")\n", + " print(\" name:\", evaluation.name)\n", + " print(\" metrics_schema_uri:\", evaluation.metrics_schema_uri)\n", + " metrics = json_format.MessageToDict(evaluation._pb.metrics)\n", + " for metric in metrics.keys():\n", + " print(metric)\n", + " print(\"logloss\", metrics[\"logLoss\"])\n", + " print(\"auPrc\", metrics[\"auPrc\"])\n", + "\n", + " return evaluation.name\n", + "\n", + "\n", + "last_evaluation = list_model_evaluations(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:automl" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created with AutoML. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"iris_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:dedicated" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `dedicated_resources`: This refers to how many compute instances (replicas) that are scaled for serving prediction requests.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `min_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:dedicated" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"iris_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " \"machine_spec\": machine_spec,\n", + " },\n", + " \"disable_container_logging\": False,\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_test_item:automl,online_prediction" + }, + "source": [ + "### Make test item\n", + "\n", + "You will use synthetic data as a test data item. Don't be concerned that we are using synthetic data -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_test_item:automl,tabular,iris" + }, + "outputs": [], + "source": [ + "INSTANCE = {\n", + " \"petal_length\": \"1.4\",\n", + " \"petal_width\": \"1.3\",\n", + " \"sepal_length\": \"5.1\",\n", + " \"sepal_width\": \"2.8\",\n", + "}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "predict_item:automl,lcn" + }, + "source": [ + "### Make a prediction\n", + "\n", + "Now you have a test item. Use this helper function `predict_item`, which takes the following parameters:\n", + "\n", + "- `filename`: The Cloud Storage path to the test item.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `parameters_dict`: Additional filtering parameters for serving prediction results.\n", + "\n", + "This function calls the prediction client service's `predict` method with the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `instances`: A list of instances (data items) to predict.\n", + "- `parameters`: Additional filtering parameters for serving prediction results. *Note*, tabular models do not support additional parameters.\n", + "\n", + "#### Request\n", + "\n", + "The format of each instance is, where values must be specified as a string:\n", + "\n", + " { 'feature_1': 'value_1', 'feature_2': 'value_2', ... }\n", + "\n", + "Since the `predict()` method can take multiple items (instances), you send your single test item as a list of one test item. As a final step, you package the instances list into Google's protobuf format -- which is what we pass to the `predict()` method.\n", + "\n", + "#### Response\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding image in the request. You will see in the output for each prediction -- in this case there is just one:\n", + "\n", + "- `confidences`: Confidence level in the prediction.\n", + "- `displayNames`: The predicted label." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "predict_item:automl,lcn" + }, + "outputs": [], + "source": [ + "def predict_item(data, endpoint, parameters_dict):\n", + " parameters = json_format.ParseDict(parameters_dict, Value())\n", + "\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [data]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", dict(prediction))\n", + "\n", + "\n", + "predict_item(INSTANCE, endpoint_id, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_automl_tabular_classification_online.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/automl/showcase_automl_tabular_classification_online_explain.ipynb b/notebooks/community/gapic/automl/showcase_automl_tabular_classification_online_explain.ipynb new file mode 100644 index 000000000..b22d7b231 --- /dev/null +++ b/notebooks/community/gapic/automl/showcase_automl_tabular_classification_online_explain.ipynb @@ -0,0 +1,1839 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: AutoML tabular classification model for online prediction with explanation\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl,xai" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to create tabular classification models and do online prediction with explanation using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:iris,lcn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Iris dataset](https://www.tensorflow.org/datasets/catalog/iris) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). This dataset does not require any feature engineering. The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts the type of Iris flower species from a class of three species: setosa, virginica, or versicolor." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,online_prediction,xai" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML tabular classification model and deploy for online prediction with explainability from a Python script using the Vertex client library. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction with explainability.\n", + "- Undeploy the Model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:v1beta1,protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:v1beta1,protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "import google.cloud.aiplatform_v1beta1 as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Set constants unique to AutoML datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the `Dataset` resource service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the `Dataset` resource service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:lcn" + }, + "outputs": [], + "source": [ + "# Tabular Dataset type\n", + "DATA_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/metadata/tables_1.0.0.yaml\"\n", + "# Tabular Labeling type\n", + "LABEL_SCHEMA = (\n", + " \"gs://google-cloud-aiplatform/schema/dataset/ioformat/table_io_format_1.0.0.yaml\"\n", + ")\n", + "# Tabular Training task\n", + "TRAINING_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_tables_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:prediction" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for prediction.\n", + "\n", + "Set the variable `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:automl" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "For AutoML batch prediction, the container image for the serving binary is pre-determined by the Vertex prediction service. More specifically, the service will pick the appropriate container for the model depending on the hardware accelerator you selected." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for prediction.\n", + "\n", + "- Set the variable `DEPLOY_COMPUTE` to configure the compute resources for the VM you will use for prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML tabular classification model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:automl,online_prediction" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training.\n", + "- Endpoint Service for deployment.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:automl,online_prediction" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:tabular" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step is to create a `Dataset` resource instance. This step differs from Vision, Video and Language. For those products, after the `Dataset` resource is created, one then separately imports the data, using the `import_data` method.\n", + "\n", + "For tabular, importing of the data is deferred until the training pipeline starts training the model. What do we do different? Well, first you won't be calling the `import_data` method. Instead, when you create the dataset instance you specify the Cloud Storage location of the CSV file or BigQuery location of the data table, which contains your tabular data as part of the `Dataset` resource's metadata.\n", + "\n", + "#### Cloud Storage\n", + "\n", + "`metadata = {\"input_config\": {\"gcs_source\": {\"uri\": [gcs_uri]}}}`\n", + "\n", + "The format for a Cloud Storage path is:\n", + "\n", + " gs://[bucket_name]/[folder(s)/[file]\n", + "\n", + "#### BigQuery\n", + "\n", + "`metadata = {\"input_config\": {\"bigquery_source\": {\"uri\": [gcs_uri]}}}`\n", + "\n", + "The format for a BigQuery path is:\n", + "\n", + " bq://[collection].[dataset].[table]\n", + "\n", + "Note that the `uri` field is a list, whereby you can input multiple CSV files or BigQuery tables when your data is split across files." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:tabular,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for tabular has a couple of requirements for your tabular data.\n", + "\n", + "- Must be in a CSV file or a BigQuery query." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:lcn,u_dataset,csv" + }, + "source": [ + "#### CSV\n", + "\n", + "For tabular classification, the CSV file has a few requirements:\n", + "\n", + "- The first row must be the heading -- note how this is different from Vision, Video and Language where the requirement is no heading.\n", + "- All but one column are features.\n", + "- One column is the label, which you will specify when you subsequently create the training pipeline." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:iris,csv,lcn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://cloud-samples-data/tables/iris_1000.csv\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:tabular" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Iris dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows.\n", + "\n", + "You also need for training to know the heading name of the label column, which is save as `label_column`. For this dataset, it is the last column in the CSV file." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:tabular" + }, + "outputs": [], + "source": [ + "count = ! gsutil cat $IMPORT_FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $IMPORT_FILE | head\n", + "\n", + "heading = ! gsutil cat $IMPORT_FILE | head -n1\n", + "label_column = str(heading).split(\",\")[-1].split(\"'\")[0]\n", + "print(\"Label Column Name\", label_column)\n", + "if label_column is None:\n", + " raise Exception(\"label column missing\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset:tabular" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + " - `metadata`: The Cloud Storage or BigQuery location of the tabular data.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset:tabular" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "\n", + "def create_dataset(name, schema, src_uri=None, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " if src_uri.startswith(\"gs://\"):\n", + " metadata = {\"input_config\": {\"gcs_source\": {\"uri\": [src_uri]}}}\n", + " elif src_uri.startswith(\"bq://\"):\n", + " metadata = {\"input_config\": {\"bigquery_source\": {\"uri\": [src_uri]}}}\n", + " dataset = aip.Dataset(\n", + " display_name=name,\n", + " metadata_schema_uri=schema,\n", + " labels=labels,\n", + " metadata=json_format.ParseDict(metadata, Value()),\n", + " )\n", + "\n", + " operation = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"iris-\" + TIMESTAMP, DATA_SCHEMA, src_uri=IMPORT_FILE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML tabular classification model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_pipeline:automl" + }, + "source": [ + "### Create a training pipeline\n", + "\n", + "You may ask, what do we use a pipeline for? You typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `pipeline_name`: A human readable name for the pipeline job.\n", + "- `model_name`: A human readable name for the model.\n", + "- `dataset`: The Vertex fully qualified dataset identifier.\n", + "- `schema`: The dataset labeling (annotation) training schema.\n", + "- `task`: A dictionary describing the requirements for the training job.\n", + "\n", + "The helper function calls the `Pipeline` client service'smethod `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "Let's look now deeper into the *minimal* requirements for constructing a `training_pipeline` specification:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The dataset labeling (annotation) training schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A human readable name for the model.\n", + "- `input_data_config`: The dataset specification.\n", + " - `dataset_id`: The Vertex dataset identifier only (non-fully qualified) -- this is the last part of the fully-qualified identifier.\n", + " - `fraction_split`: If specified, the percentages of the dataset to use for training, test and validation. Otherwise, the percentages are automatically selected by AutoML." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_pipeline:automl" + }, + "outputs": [], + "source": [ + "def create_pipeline(pipeline_name, model_name, dataset, schema, task):\n", + "\n", + " dataset_id = dataset.split(\"/\")[-1]\n", + "\n", + " input_config = {\n", + " \"dataset_id\": dataset_id,\n", + " \"fraction_split\": {\n", + " \"training_fraction\": 0.8,\n", + " \"validation_fraction\": 0.1,\n", + " \"test_fraction\": 0.1,\n", + " },\n", + " }\n", + "\n", + " training_pipeline = {\n", + " \"display_name\": pipeline_name,\n", + " \"training_task_definition\": schema,\n", + " \"training_task_inputs\": task,\n", + " \"input_data_config\": input_config,\n", + " \"model_to_upload\": {\"display_name\": model_name},\n", + " }\n", + "\n", + " try:\n", + " pipeline = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " )\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "task_requirements:automl,tabular" + }, + "source": [ + "### Construct the task requirements\n", + "\n", + "Next, construct the task requirements. Unlike other parameters which take a Python (JSON-like) dictionary, the `task` field takes a Google protobuf Struct, which is very similar to a Python dictionary. Use the `json_format.ParseDict` method for the conversion.\n", + "\n", + "The minimal fields you need to specify are:\n", + "\n", + "- `prediction_type`: Whether we are doing \"classification\" or \"regression\".\n", + "- `target_column`: The CSV heading column name for the column we want to predict (i.e., the label).\n", + "- `train_budget_milli_node_hours`: The maximum time to budget (billed) for training the model, where 1000 = 1 hour.\n", + "- `disable_early_stopping`: Whether True/False to let AutoML use its judgement to stop training early or train for the entire budget.\n", + "- `transformations`: Specifies the feature engineering for each feature column.\n", + "\n", + "For `transformations`, the list must have an entry for each column. The outer key field indicates the type of feature engineering for the corresponding column. In this tutorial, you set it to `\"auto\"` to tell AutoML to automatically determine it.\n", + "\n", + "Finally, create the pipeline by calling the helper function `create_pipeline`, which returns an instance of a training pipeline object." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_transformations:automl,tabular,iris" + }, + "outputs": [], + "source": [ + "TRANSFORMATIONS = [\n", + " {\"auto\": {\"column_name\": \"sepal_width\"}},\n", + " {\"auto\": {\"column_name\": \"sepal_length\"}},\n", + " {\"auto\": {\"column_name\": \"petal_length\"}},\n", + " {\"auto\": {\"column_name\": \"petal_width\"}},\n", + "]" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_requirements:automl,tabular,transformations" + }, + "outputs": [], + "source": [ + "PIPE_NAME = \"iris_pipe-\" + TIMESTAMP\n", + "MODEL_NAME = \"iris_model-\" + TIMESTAMP\n", + "\n", + "task = Value(\n", + " struct_value=Struct(\n", + " fields={\n", + " \"target_column\": Value(string_value=label_column),\n", + " \"prediction_type\": Value(string_value=\"classification\"),\n", + " \"train_budget_milli_node_hours\": Value(number_value=1000),\n", + " \"disable_early_stopping\": Value(bool_value=False),\n", + " \"transformations\": json_format.ParseDict(TRANSFORMATIONS, Value()),\n", + " }\n", + " )\n", + ")\n", + "\n", + "response = create_pipeline(PIPE_NAME, MODEL_NAME, dataset_id, TRAINING_SCHEMA, task)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 30 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "model_information" + }, + "source": [ + "## Model information\n", + "\n", + "Now that your model is trained, you can get some information on your model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:automl" + }, + "source": [ + "## Evaluate the Model resource\n", + "\n", + "Now find out how good the model service believes your model is. As part of training, some portion of the dataset was set aside as the test (holdout) data, which is used by the pipeline service to evaluate the model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "list_model_evaluations:automl,lcn" + }, + "source": [ + "### List evaluations for all slices\n", + "\n", + "Use this helper function `list_model_evaluations`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified model identifier for the `Model` resource.\n", + "\n", + "This helper function uses the model client service's `list_model_evaluations` method, which takes the same parameter. The response object from the call is a list, where each element is an evaluation metric.\n", + "\n", + "For each evaluation (you probably only have one) we then print all the key names for each metric in the evaluation, and for a small set (`logLoss` and `auPrc`) you will print the result." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "list_model_evaluations:automl,lcn" + }, + "outputs": [], + "source": [ + "def list_model_evaluations(name):\n", + " response = clients[\"model\"].list_model_evaluations(parent=name)\n", + " for evaluation in response:\n", + " print(\"model_evaluation\")\n", + " print(\" name:\", evaluation.name)\n", + " print(\" metrics_schema_uri:\", evaluation.metrics_schema_uri)\n", + " metrics = json_format.MessageToDict(evaluation._pb.metrics)\n", + " for metric in metrics.keys():\n", + " print(metric)\n", + " print(\"logloss\", metrics[\"logLoss\"])\n", + " print(\"auPrc\", metrics[\"auPrc\"])\n", + "\n", + " return evaluation.name\n", + "\n", + "\n", + "last_evaluation = list_model_evaluations(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:automl" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created with AutoML. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"iris_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:dedicated,v1beta1" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the model to the endpoint you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified identifier of the `Model` resource to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to to.\n", + "- `deployed_model`: The requirements for deploying the model.\n", + "- `traffic_split`: Percent of traffic at endpoint that goes to this model, which is specified as a dictioney of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then specify as, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + " { \"0\": percent, model_id: percent, ... }\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified identifier of the (upload) `Model` resource to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `dedicated_resources`: This refers to how many compute instances (replicas) that are scaled for serving prediction requests.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `min_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "- `enable_container_logging`: This enables logging of container events, such as execution failures (default is container logging is disabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:dedicated,v1beta1" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"iris_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " \"machine_spec\": machine_spec,\n", + " },\n", + " \"enable_container_logging\": False,\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction:xai" + }, + "source": [ + "## Make a online prediction request with explainability\n", + "\n", + "Now do a online prediction with explainability to your deployed model. In this method, the predicted response will include an explanation on how the features contributed to the explanation." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_test_item:automl,online_prediction" + }, + "source": [ + "### Make test item\n", + "\n", + "You will use synthetic data as a test data item. Don't be concerned that we are using synthetic data -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_test_item:automl,tabular,iris" + }, + "outputs": [], + "source": [ + "INSTANCE = {\n", + " \"petal_length\": \"1.4\",\n", + " \"petal_width\": \"1.3\",\n", + " \"sepal_length\": \"5.1\",\n", + " \"sepal_width\": \"2.8\",\n", + "}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "explain_item:automl,lcn" + }, + "source": [ + "### Make a prediction with explanation\n", + "\n", + "Ok, now you have a test item. Use this helper function `explain_item`, which takes the following parameters:\n", + "\n", + "- `data_items`: The test tabular data items.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `parameters_dict`: Additional filtering parameters for serving prediction results -- in your case you will pass `None`.\n", + "\n", + "This function uses the prediction client service and calls the `explain` method with the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `instances`: A list of instances (data items) to predict.\n", + "- `parameters`: Additional filtering parameters for serving prediction results. *Note*, image segmentation models do not support additional parameters.\n", + "- `deployed_model_id`: The Vertex fully qualified identifier for the deployed model, when more than one model is deployed at the endpoint. Otherwise, if only one model deployed, can be set to `None`.\n", + "\n", + "#### Request\n", + "\n", + "The format of each instance is:\n", + "\n", + " { 'content': text_item }\n", + "\n", + "Since the `explain()` method can take multiple items (instances), you send your single test item as a list of one test item. As a final step, you package the instances list into Google's protobuf format -- which is what we pass to the `explain()` method.\n", + "\n", + "#### Response\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding image in the request. You will see in the output for each prediction -- in this case there is just one:\n", + "\n", + "- `deployed_model_id` -- The Vertex fully qualified identifer for the `Model` resource that did the prediction/explanation.\n", + "- `predictions` -- The predicated class and confidence level between 0 and 1.\n", + " - `confidences`: Confidence level in the prediction.\n", + " - `displayNames`: The predicted label.\n", + "- `explanations` -- How each feature contributed to the prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "explain_item:automl,lcn" + }, + "outputs": [], + "source": [ + "def explain_item(\n", + " data_items, endpoint, parameters_dict, deployed_model_id, silent=False\n", + "):\n", + " parameters = json_format.ParseDict(parameters_dict, Value())\n", + "\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances = [json_format.ParseDict(s, Value()) for s in data_items]\n", + "\n", + " response = clients[\"prediction\"].explain(\n", + " endpoint=endpoint,\n", + " instances=instances,\n", + " parameters=parameters,\n", + " deployed_model_id=deployed_model_id,\n", + " )\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " try:\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", dict(prediction))\n", + " except:\n", + " pass\n", + "\n", + " explanations = response.explanations\n", + " print(\"explanations\")\n", + " for explanation in explanations:\n", + " print(explanation)\n", + " return response\n", + "\n", + "\n", + "response = explain_item([INSTANCE], endpoint_id, None, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "understand_explanations" + }, + "source": [ + "### Understanding the explanations response\n", + "\n", + "First, you will look what your model predicted and compare it to the actual value." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "understand_explanations:lcn" + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "\n", + "try:\n", + " predictions = response.predictions\n", + " label = np.argmax(predictions[0][\"scores\"])\n", + " cls = predictions[0][\"classes\"][label]\n", + " print(\"Predicted Value:\", cls, predictions[0][\"scores\"][label])\n", + "except:\n", + " pass" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "examine_feature_attributions" + }, + "source": [ + "### Examine feature attributions\n", + "\n", + "Next you will look at the feature attributions for this particular example. Positive attribution values mean a particular feature pushed your model prediction up by that amount, and vice versa for negative attribution values." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "examine_feature_attributions:iris" + }, + "outputs": [], + "source": [ + "from tabulate import tabulate\n", + "\n", + "feature_names = [\"petal_length\", \"petal_width\", \"sepal_length\", \"sepal_width\"]\n", + "attributions = response.explanations[0].attributions[0].feature_attributions\n", + "\n", + "rows = []\n", + "for i, val in enumerate(feature_names):\n", + " rows.append([val, INSTANCE[val], attributions[val]])\n", + "print(tabulate(rows, headers=[\"Feature name\", \"Feature value\", \"Attribution value\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "check_explanations_baselines" + }, + "source": [ + "### Check your explanations and baselines\n", + "\n", + "To better make sense of the feature attributions you're getting, you should compare them with your model's baseline. In most cases, the sum of your attribution values + the baseline should be very close to your model's predicted value for each input. Also note that for regression models, the `baseline_score` returned from AI Explanations will be the same for each example sent to your model. For classification models, each class will have its own baseline.\n", + "\n", + "In this section you'll send 10 test examples to your model for prediction in order to compare the feature attributions with the baseline. Then you'll run each test example's attributions through a sanity check in the `sanity_check_explanations` method.\n", + "\n", + "#### Get explanations" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "check_explanations_baselines:iris" + }, + "outputs": [], + "source": [ + "import random\n", + "\n", + "# Prepare 10 test examples to your model for prediction using a random distribution to generate\n", + "# test instances\n", + "instances = []\n", + "for i in range(10):\n", + " pl = str(random.uniform(1.0, 2.0))\n", + " pw = str(random.uniform(1.0, 2.0))\n", + " sl = str(random.uniform(4.0, 6.0))\n", + " sw = str(random.uniform(2.0, 4.0))\n", + " instances.append(\n", + " {\"petal_length\": pl, \"petal_width\": pw, \"sepal_length\": sl, \"sepal_width\": sw}\n", + " )\n", + "\n", + "response = explain_item(instances, endpoint_id, None, None, silent=True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sanity_check_explanations" + }, + "source": [ + "#### Sanity check\n", + "\n", + "In the function below you perform a sanity check on the explanations." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "sanity_check_explanations" + }, + "outputs": [], + "source": [ + "def sanity_check_explanations(\n", + " explanation, prediction, mean_tgt_value=None, variance_tgt_value=None\n", + "):\n", + " passed_test = 0\n", + " total_test = 1\n", + " # `attributions` is a dict where keys are the feature names\n", + " # and values are the feature attributions for each feature\n", + " baseline_score = explanation.attributions[0].baseline_output_value\n", + " print(\"baseline:\", baseline_score)\n", + "\n", + " # Sanity check 1\n", + " # The prediction at the input is equal to that at the baseline.\n", + " # Please use a different baseline. Some suggestions are: random input, training\n", + " # set mean.\n", + " if abs(prediction - baseline_score) <= 0.05:\n", + " print(\"Warning: example score and baseline score are too close.\")\n", + " print(\"You might not get attributions.\")\n", + " else:\n", + " passed_test += 1\n", + " print(\"Sanity Check 1: Passed\")\n", + "\n", + " print(passed_test, \" out of \", total_test, \" sanity checks passed.\")\n", + "\n", + "\n", + "i = 0\n", + "for explanation in response.explanations:\n", + " try:\n", + " prediction = np.max(response.predictions[i][\"scores\"])\n", + " except TypeError:\n", + " prediction = np.max(response.predictions[i])\n", + " sanity_check_explanations(explanation, prediction)\n", + " i += 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_automl_tabular_classification_online_explain.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/automl/showcase_automl_tabular_forecasting_batch.ipynb b/notebooks/community/gapic/automl/showcase_automl_tabular_forecasting_batch.ipynb new file mode 100644 index 000000000..33f1f5181 --- /dev/null +++ b/notebooks/community/gapic/automl/showcase_automl_tabular_forecasting_batch.ipynb @@ -0,0 +1,1761 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: AutoML tabular forecasting model for batch prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to create tabular forecasting models and do batch prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:covid,forecast" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the BigQuery public dataset, [The New York Times US Coronavirus Database](https://pantheon.corp.google.com/marketplace/product/the-new-york-times/covid19_us_cases). This dataset does not require any feature engineering. The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,batch_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML tabular forecasting model from a Python script, and then do a batch prediction using the Vertex client library. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Make a batch prediction.\n", + "\n", + "There is one key difference between using batch prediction and using online prediction:\n", + "\n", + "* Prediction Service: Does an on-demand prediction for the entire set of instances (i.e., one or more data items) and returns the results in real-time.\n", + "\n", + "* Batch Prediction Service: Does a queued (batch) prediction for the entire set of instances in the background and stores the results in a Cloud Storage bucket when ready." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Set constants unique to AutoML datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the `Dataset` resource service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the `Dataset` resource service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:forecast" + }, + "outputs": [], + "source": [ + "# Forecasting Dataset type\n", + "DATA_SCHEMA = (\n", + " \"gs://google-cloud-aiplatform/schema/dataset/metadata/time_series_1.0.0.yaml\"\n", + ")\n", + "# Forecasting Training task\n", + "TRAINING_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_forecasting_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:prediction" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for prediction.\n", + "\n", + "Set the variable `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:automl" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "For AutoML batch prediction, the container image for the serving binary is pre-determined by the Vertex prediction service. More specifically, the service will pick the appropriate container for the model depending on the hardware accelerator you selected." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for prediction.\n", + "\n", + "- Set the variable `DEPLOY_COMPUTE` to configure the compute resources for the VM you will use for prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML tabular forecasting model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:automl,batch_prediction" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training.\n", + "- Job Service for batch prediction and custom training." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:automl,batch_prediction" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"job\"] = create_job_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:tabular,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for tabular has a couple of requirements for your tabular data.\n", + "\n", + "- Must be in a CSV file or a BigQuery query." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:forecast,u_dataset,csv" + }, + "source": [ + "#### CSV\n", + "\n", + "For tabular forecasting, the CSV file has a few requirements:\n", + "\n", + "- The first row must be the heading -- note how this is different from Vision, Video and Language where the requirement is no heading.\n", + "- All but one column are features.\n", + "- One column is the label, which you will specify when you subsequently create the training pipeline.\n", + "- One column is the time column, which you will specify when you subsequently create the training pipeline.\n", + "- One column is the time series identifier column, which you will specify when you subsequently create the training pipeline." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:covid,csv,forecast" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://cloud-samples-data/ai-platform/covid/bigquery-public-covid-nyt-us-counties-train.csv\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:forecast" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the NY Times COVID Database dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows.\n", + "\n", + "You will also need for training to know the heading name of the label column, which we will save as `label_column`. Moreover, to identify the time series in the file, the training needs to know two additional columns, `time_column` that provides temporal information and `time_series_identifier_column` in which the cells with the same value consititue one time series. For this dataset, the label column is `deaths`, the time column is `date` and the time series identifier column is `county`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:forecast" + }, + "outputs": [], + "source": [ + "count = ! gsutil cat $IMPORT_FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $IMPORT_FILE | head\n", + "\n", + "heading = ! gsutil cat $IMPORT_FILE | head -n1\n", + "label_column = \"deaths\" # @param {type:\"string\"}\n", + "time_column = \"date\" # @param {type:\"string\"}\n", + "time_series_identifier_column = \"county\" # @param {type:\"string\"}\n", + "print(\"Label Column Name\", label_column)\n", + "print(\"Time Column Name\", time_column)\n", + "print(\"Time Series Identifier Column Name\", time_series_identifier_column)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset:tabular" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + " - `metadata`: The Cloud Storage or BigQuery location of the tabular data.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset:tabular" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "\n", + "def create_dataset(name, schema, src_uri=None, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " if src_uri.startswith(\"gs://\"):\n", + " metadata = {\"input_config\": {\"gcs_source\": {\"uri\": [src_uri]}}}\n", + " elif src_uri.startswith(\"bq://\"):\n", + " metadata = {\"input_config\": {\"bigquery_source\": {\"uri\": [src_uri]}}}\n", + " dataset = aip.Dataset(\n", + " display_name=name,\n", + " metadata_schema_uri=schema,\n", + " labels=labels,\n", + " metadata=json_format.ParseDict(metadata, Value()),\n", + " )\n", + "\n", + " operation = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"covid-\" + TIMESTAMP, DATA_SCHEMA, src_uri=IMPORT_FILE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML tabular forecasting model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_pipeline:automl" + }, + "source": [ + "### Create a training pipeline\n", + "\n", + "You may ask, what do we use a pipeline for? You typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `pipeline_name`: A human readable name for the pipeline job.\n", + "- `model_name`: A human readable name for the model.\n", + "- `dataset`: The Vertex fully qualified dataset identifier.\n", + "- `schema`: The dataset labeling (annotation) training schema.\n", + "- `task`: A dictionary describing the requirements for the training job.\n", + "\n", + "The helper function calls the `Pipeline` client service'smethod `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "Let's look now deeper into the *minimal* requirements for constructing a `training_pipeline` specification:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The dataset labeling (annotation) training schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A human readable name for the model.\n", + "- `input_data_config`: The dataset specification.\n", + " - `dataset_id`: The Vertex dataset identifier only (non-fully qualified) -- this is the last part of the fully-qualified identifier.\n", + " - `fraction_split`: If specified, the percentages of the dataset to use for training, test and validation. Otherwise, the percentages are automatically selected by AutoML." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_pipeline:automl" + }, + "outputs": [], + "source": [ + "def create_pipeline(pipeline_name, model_name, dataset, schema, task):\n", + "\n", + " dataset_id = dataset.split(\"/\")[-1]\n", + "\n", + " input_config = {\n", + " \"dataset_id\": dataset_id,\n", + " \"fraction_split\": {\n", + " \"training_fraction\": 0.8,\n", + " \"validation_fraction\": 0.1,\n", + " \"test_fraction\": 0.1,\n", + " },\n", + " }\n", + "\n", + " training_pipeline = {\n", + " \"display_name\": pipeline_name,\n", + " \"training_task_definition\": schema,\n", + " \"training_task_inputs\": task,\n", + " \"input_data_config\": input_config,\n", + " \"model_to_upload\": {\"display_name\": model_name},\n", + " }\n", + "\n", + " try:\n", + " pipeline = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " )\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "task_requirements:automl,forecast" + }, + "source": [ + "### Construct the task requirements\n", + "\n", + "Next, construct the task requirements. Unlike other parameters which take a Python (JSON-like) dictionary, the `task` field takes a Google protobuf Struct, which is very similar to a Python dictionary. Use the `json_format.ParseDict` method for the conversion.\n", + "\n", + "The minimal fields you need to specify are:\n", + "\n", + "- `target_column`: The CSV heading column name for the column we want to predict (i.e., the label).\n", + "- `time_column`: The CSV heading column name for the column that represents the temporal information of time series.\n", + "- `time_series_identifier_column`: The CSV heading column name for the column whose value identifies individual time series, e.g. rows of the value at this column consistues one time series.\n", + "- `period`: the time interval between two consecutive observations in the same time series, e.g. hourly, daily etc.\n", + "- `forecast_window_end`: the number to indicate how many `period` in future you want to forecast.\n", + "- `time_variant_past_only_columns`: The CSV heading column names for the columns that not known at forecast time, (i.e., the label).\n", + "- `train_budget_milli_node_hours`: The maximum time to budget (billed) for training the model, where 1000 = 1 hour.\n", + "- `disable_early_stopping`: Whether True/False to let AutoML use its judgement to stop training early or train for the entire budget.\n", + "- `transformations`: Specifies the feature engineering for each feature column.\n", + "\n", + "For `transformations`, the list must have an entry for each column. The outer key field indicates the type of feature engineering for the corresponding column. In this tutorial, you set it to `\"auto\"` to tell AutoML to automatically determine it.\n", + "\n", + "Finally, create the pipeline by calling the helper function `create_pipeline`, which returns an instance of a training pipeline object." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_transformations:automl,forecast,covid" + }, + "outputs": [], + "source": [ + "TRANSFORMATIONS = [\n", + " {\"auto\": {\"column_name\": \"date\"}},\n", + " {\"auto\": {\"column_name\": \"state_name\"}},\n", + " {\"auto\": {\"column_name\": \"county_fips_code\"}},\n", + " {\"auto\": {\"column_name\": \"confirmed_cases\"}},\n", + " {\"auto\": {\"column_name\": \"deaths\"}},\n", + "]\n", + "\n", + "PERIOD = {\"unit\": \"day\", \"quantity\": 1}\n", + "\n", + "PAST_ONLY_COLUMNS = [\"deaths\"]\n", + "STATIC_COLUMNS = [\"state_name\"]\n", + "PAST_AND_FUTURE_COLUMNS = [\"date\"]\n", + "\n", + "optimization_objective = \"minimize-rmse\"" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_requirements:automl,forecast,transformations" + }, + "outputs": [], + "source": [ + "PIPE_NAME = \"covid_pipe-\" + TIMESTAMP\n", + "MODEL_NAME = \"covid_model-\" + TIMESTAMP\n", + "\n", + "task = Value(\n", + " struct_value=Struct(\n", + " fields={\n", + " \"target_column\": Value(string_value=label_column),\n", + " \"time_column\": Value(string_value=time_column),\n", + " \"time_series_identifier_column\": Value(\n", + " string_value=time_series_identifier_column\n", + " ),\n", + " \"period\": json_format.ParseDict(PERIOD, Value()),\n", + " \"forecast_window_end\": Value(number_value=10),\n", + " \"time_variant_past_only_columns\": json_format.ParseDict(\n", + " PAST_ONLY_COLUMNS, Value()\n", + " ),\n", + " \"static_columns\": json_format.ParseDict(STATIC_COLUMNS, Value()),\n", + " \"time_variant_past_and_future_columns\": json_format.ParseDict(\n", + " PAST_AND_FUTURE_COLUMNS, Value()\n", + " ),\n", + " \"train_budget_milli_node_hours\": Value(number_value=1000),\n", + " \"optimization_objective\": Value(string_value=optimization_objective),\n", + " \"transformations\": json_format.ParseDict(TRANSFORMATIONS, Value()),\n", + " }\n", + " )\n", + ")\n", + "response = create_pipeline(PIPE_NAME, MODEL_NAME, dataset_id, TRAINING_SCHEMA, task)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 60 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "model_information" + }, + "source": [ + "## Model information\n", + "\n", + "Now that your model is trained, you can get some information on your model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:automl" + }, + "source": [ + "## Evaluate the Model resource\n", + "\n", + "Now find out how good the model service believes your model is. As part of training, some portion of the dataset was set aside as the test (holdout) data, which is used by the pipeline service to evaluate the model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "list_model_evaluations:automl,forecast" + }, + "source": [ + "### List evaluations for all slices\n", + "\n", + "Use this helper function `list_model_evaluations`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified model identifier for the `Model` resource.\n", + "\n", + "This helper function uses the model client service's `list_model_evaluations` method, which takes the same parameter. The response object from the call is a list, where each element is an evaluation metric.\n", + "\n", + "For each evaluation -- you probably only have one, we then print all the key names for each metric in the evaluation, and for a small set (`rootMeanSquaredError` and `meanAbsoluteError`) you will print the result." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "list_model_evaluations:automl,forecast" + }, + "outputs": [], + "source": [ + "def list_model_evaluations(name):\n", + " response = clients[\"model\"].list_model_evaluations(parent=name)\n", + " for evaluation in response:\n", + " print(\"model_evaluation\")\n", + " print(\" name:\", evaluation.name)\n", + " print(\" metrics_schema_uri:\", evaluation.metrics_schema_uri)\n", + " metrics = json_format.MessageToDict(evaluation._pb.metrics)\n", + " for metric in metrics.keys():\n", + " print(metric)\n", + " print(\"rootMeanSquaredError\", metrics[\"rootMeanSquaredError\"])\n", + " print(\"meanAbsoluteError\", metrics[\"meanAbsoluteError\"])\n", + "\n", + " return evaluation.name\n", + "\n", + "\n", + "last_evaluation = list_model_evaluations(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy:batch_prediction" + }, + "source": [ + "## Model deployment for batch prediction\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created for batch prediction. This differs from deploying a `Model` resource for on-demand prediction.\n", + "\n", + "For online prediction, you:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource.\n", + "\n", + "3. Make online prediction requests to the `Endpoint` resource.\n", + "\n", + "For batch-prediction, you:\n", + "\n", + "1. Create a batch prediction job.\n", + "\n", + "2. The job service will provision resources for the batch prediction request.\n", + "\n", + "3. The results of the batch prediction request are returned to the caller.\n", + "\n", + "4. The job service will unprovision the resoures for the batch prediction request." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a batch prediction request\n", + "\n", + "Now do a batch prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_test_items:automl,batch_prediction" + }, + "source": [ + "### Make test items\n", + "\n", + "You will use synthetic data as a test data items. Don't be concerned that we are using synthetic data -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_test_items:automl,tabular,covid" + }, + "outputs": [], + "source": [ + "HEADING = \"date,county,state_name,county_fips_code,confirmed_cases,deaths\"\n", + "INSTANCE_1 = \"2020-10-13,Adair,Iowa,19001,103,null\"\n", + "INSTANCE_2 = \"2020-10-29,Adair,Iowa,19001,197,null\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:automl,tabular" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "Now make a batch input file, which you will store in your local Cloud Storage bucket. Unlike image, video and text, the batch input file for tabular is only supported for CSV. For CSV file, you make:\n", + "\n", + "- The first line is the heading with the feature (fields) heading names.\n", + "- Each remaining line is a separate prediction request with the corresponding feature values.\n", + "\n", + "For example:\n", + "\n", + " \"feature_1\", \"feature_2\". ...\n", + " value_1, value_2, ..." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_file:automl,tabular" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "gcs_input_uri = BUCKET_NAME + \"/test.csv\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " f.write(HEADING + \"\\n\")\n", + " f.write(str(INSTANCE_1) + \"\\n\")\n", + " f.write(str(INSTANCE_2) + \"\\n\")\n", + "\n", + "print(gcs_input_uri)\n", + "! gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your batch prediction requests:\n", + "\n", + "- Single Instance: The batch prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The batch prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and batch prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The batch prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_request:automl,forecast" + }, + "source": [ + "### Make batch prediction request\n", + "\n", + "Now that your batch of two test items is ready, let's do the batch request. Use this helper function `create_batch_prediction_job`, with the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the prediction job.\n", + "- `model_name`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `gcs_source_uri`: The Cloud Storage path to the input file -- which you created above.\n", + "- `gcs_destination_output_uri_prefix`: The Cloud Storage path that the service will write the predictions to.\n", + "- `parameters`: Additional filtering parameters for serving prediction results.\n", + "\n", + "The helper function calls the job client service's `create_batch_prediction_job` metho, with the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for Dataset, Model and Pipeline resources.\n", + "- `batch_prediction_job`: The specification for the batch prediction job.\n", + "\n", + "Let's now dive into the specification for the `batch_prediction_job`:\n", + "\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `dedicated_resources`: The compute resources to provision for the batch prediction job.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `starting_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "- `model_parameters`: Additional filtering parameters for serving prediction results. *Note*, image segmentation models do not support additional parameters.\n", + "- `input_config`: The input source and format type for the instances to predict.\n", + " - `instances_format`: The format of the batch prediction request file: `csv` only supported.\n", + " - `gcs_source`: A list of one or more Cloud Storage paths to your batch prediction requests.\n", + "- `output_config`: The output destination and format for the predictions.\n", + " - `prediction_format`: The format of the batch prediction response file: `csv` only supported.\n", + " - `gcs_destination`: The output destination for the predictions.\n", + "\n", + "This call is an asychronous operation. You will print from the response object a few select fields, including:\n", + "\n", + "- `name`: The Vertex fully qualified identifier assigned to the batch prediction job.\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the Model resource.\n", + "- `generate_explanations`: Whether True/False explanations were provided with the predictions (explainability).\n", + "- `state`: The state of the prediction job (pending, running, etc).\n", + "\n", + "Since this call will take a few moments to execute, you will likely get `JobState.JOB_STATE_PENDING` for `state`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_request:automl,forecast" + }, + "outputs": [], + "source": [ + "BATCH_MODEL = \"covid_batch-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_batch_prediction_job(\n", + " display_name,\n", + " model_name,\n", + " gcs_source_uri,\n", + " gcs_destination_output_uri_prefix,\n", + " parameters=None,\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " batch_prediction_job = {\n", + " \"display_name\": display_name,\n", + " # Format: 'projects/{project}/locations/{location}/models/{model_id}'\n", + " \"model\": model_name,\n", + " \"model_parameters\": json_format.ParseDict(parameters, Value()),\n", + " \"input_config\": {\n", + " \"instances_format\": IN_FORMAT,\n", + " \"gcs_source\": {\"uris\": [gcs_source_uri]},\n", + " },\n", + " \"output_config\": {\n", + " \"predictions_format\": OUT_FORMAT,\n", + " \"gcs_destination\": {\"output_uri_prefix\": gcs_destination_output_uri_prefix},\n", + " },\n", + " \"dedicated_resources\": {\n", + " \"machine_spec\": machine_spec,\n", + " \"starting_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " },\n", + " }\n", + " response = clients[\"job\"].create_batch_prediction_job(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + " )\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try:\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", response.labels)\n", + " return response\n", + "\n", + "\n", + "IN_FORMAT = \"csv\"\n", + "OUT_FORMAT = \"csv\" # [csv]\n", + "\n", + "response = create_batch_prediction_job(\n", + " BATCH_MODEL, model_to_deploy_id, gcs_input_uri, BUCKET_NAME, None\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batch_job_id:response" + }, + "source": [ + "Now get the unique identifier for the batch prediction job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batch_job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the batch job\n", + "batch_job_id = response.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split(\"/\")[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_batch_prediction_job" + }, + "source": [ + "### Get information on a batch prediction job\n", + "\n", + "Use this helper function `get_batch_prediction_job`, with the following paramter:\n", + "\n", + "- `job_name`: The Vertex fully qualified identifier for the batch prediction job.\n", + "\n", + "The helper function calls the job client service's `get_batch_prediction_job` method, with the following paramter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the batch prediction job. In this tutorial, you will pass it the Vertex fully qualified identifier for your batch prediction job -- `batch_job_id`\n", + "\n", + "The helper function will return the Cloud Storage path to where the predictions are stored -- `gcs_destination`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_batch_prediction_job" + }, + "outputs": [], + "source": [ + "def get_batch_prediction_job(job_name, silent=False):\n", + " response = clients[\"job\"].get_batch_prediction_job(name=job_name)\n", + " if silent:\n", + " return response.output_config.gcs_destination.output_uri_prefix, response.state\n", + "\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try: # not all data types support explanations\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" error:\", response.error)\n", + " gcs_destination = response.output_config.gcs_destination\n", + " print(\" gcs_destination\")\n", + " print(\" output_uri_prefix:\", gcs_destination.output_uri_prefix)\n", + " return gcs_destination.output_uri_prefix, response.state\n", + "\n", + "\n", + "predictions, state = get_batch_prediction_job(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_the_predictions:automl,forecast" + }, + "source": [ + "### Get Predictions\n", + "\n", + "When the batch prediction is done processing, the job state will be `JOB_STATE_SUCCEEDED`.\n", + "\n", + "Finally you view the predictions stored at the Cloud Storage path you set as output. The predictions will be in a CSV format, which you indicated at the time you made the batch prediction job, under a subfolder starting with the name `prediction`, and under that folder will be a file called `predictions*.csv`.\n", + "\n", + "Now display (cat) the contents. You will see multiple rows, one for each prediction.\n", + "\n", + "For each prediction:\n", + "\n", + "- The first four fields are the values (features) you did the prediction on.\n", + "- The remaining fields are the confidence values, between 0 and 1, for each prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_the_predictions:automl,tabular" + }, + "outputs": [], + "source": [ + "def get_latest_predictions(gcs_out_dir):\n", + " \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n", + " folders = !gsutil ls $gcs_out_dir\n", + " latest = \"\"\n", + " for folder in folders:\n", + " subfolder = folder.split(\"/\")[-2]\n", + " if subfolder.startswith(\"prediction-\"):\n", + " if subfolder > latest:\n", + " latest = folder[:-1]\n", + " return latest\n", + "\n", + "\n", + "while True:\n", + " predictions, state = get_batch_prediction_job(batch_job_id, True)\n", + " if state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", state)\n", + " if state == aip.JobState.JOB_STATE_FAILED:\n", + " raise Exception(\"Batch Job Failed\")\n", + " else:\n", + " folder = get_latest_predictions(predictions)\n", + " ! gsutil ls $folder/prediction*.csv\n", + "\n", + " ! gsutil cat $folder/prediction*.csv\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_automl_tabular_forecasting_batch.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/automl/showcase_automl_tabular_regression_online_bq.ipynb b/notebooks/community/gapic/automl/showcase_automl_tabular_regression_online_bq.ipynb new file mode 100644 index 000000000..63469775a --- /dev/null +++ b/notebooks/community/gapic/automl/showcase_automl_tabular_regression_online_bq.ipynb @@ -0,0 +1,1633 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: AutoML tabular regression model for online prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to create tabular regression models and do online prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:gsod,lrg" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the GSOD dataset from [BigQuery public datasets](https://cloud.google.com/bigquery/public-data). The version of the dataset you will use only the fields year, month and day to predict the value of mean daily temperature (mean_temp)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,online_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML tabular regression model and deploy for online prediction from a Python script using the Vertex client library. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model`." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Set constants unique to AutoML datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the `Dataset` resource service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the `Dataset` resource service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:lrg" + }, + "outputs": [], + "source": [ + "# Tabular Dataset type\n", + "DATA_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/metadata/tables_1.0.0.yaml\"\n", + "# Tabular Labeling type\n", + "LABEL_SCHEMA = (\n", + " \"gs://google-cloud-aiplatform/schema/dataset/ioformat/table_io_format_1.0.0.yaml\"\n", + ")\n", + "# Tabular Training task\n", + "TRAINING_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_tables_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:prediction" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for prediction.\n", + "\n", + "Set the variable `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:automl" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "For AutoML batch prediction, the container image for the serving binary is pre-determined by the Vertex prediction service. More specifically, the service will pick the appropriate container for the model depending on the hardware accelerator you selected." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for prediction.\n", + "\n", + "- Set the variable `DEPLOY_COMPUTE` to configure the compute resources for the VM you will use for prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML tabular regression model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:automl,online_prediction" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training.\n", + "- Endpoint Service for deployment.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:automl,online_prediction" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:tabular" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step is to create a `Dataset` resource instance. This step differs from Vision, Video and Language. For those products, after the `Dataset` resource is created, one then separately imports the data, using the `import_data` method.\n", + "\n", + "For tabular, importing of the data is deferred until the training pipeline starts training the model. What do we do different? Well, first you won't be calling the `import_data` method. Instead, when you create the dataset instance you specify the Cloud Storage location of the CSV file or BigQuery location of the data table, which contains your tabular data as part of the `Dataset` resource's metadata.\n", + "\n", + "#### Cloud Storage\n", + "\n", + "`metadata = {\"input_config\": {\"gcs_source\": {\"uri\": [gcs_uri]}}}`\n", + "\n", + "The format for a Cloud Storage path is:\n", + "\n", + " gs://[bucket_name]/[folder(s)/[file]\n", + "\n", + "#### BigQuery\n", + "\n", + "`metadata = {\"input_config\": {\"bigquery_source\": {\"uri\": [gcs_uri]}}}`\n", + "\n", + "The format for a BigQuery path is:\n", + "\n", + " bq://[collection].[dataset].[table]\n", + "\n", + "Note that the `uri` field is a list, whereby you can input multiple CSV files or BigQuery tables when your data is split across files." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:tabular,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for tabular has a couple of requirements for your tabular data.\n", + "\n", + "- Must be in a CSV file or a BigQuery query." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,bq" + }, + "source": [ + "#### Location of BigQuery training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the data table in BigQuery." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:gsod,bq,lrg" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"bq://bigquery-public-data.samples.gsod\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:bq" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the NOAA historical weather data dataset that is stored in a public BigQuery table.\n", + "\n", + "**TODO**\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the table (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:bq" + }, + "outputs": [], + "source": [ + "!bq head -n 10 $IMPORT_FILE" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset:tabular" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + " - `metadata`: The Cloud Storage or BigQuery location of the tabular data.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset:tabular" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "\n", + "def create_dataset(name, schema, src_uri=None, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " if src_uri.startswith(\"gs://\"):\n", + " metadata = {\"input_config\": {\"gcs_source\": {\"uri\": [src_uri]}}}\n", + " elif src_uri.startswith(\"bq://\"):\n", + " metadata = {\"input_config\": {\"bigquery_source\": {\"uri\": [src_uri]}}}\n", + " dataset = aip.Dataset(\n", + " display_name=name,\n", + " metadata_schema_uri=schema,\n", + " labels=labels,\n", + " metadata=json_format.ParseDict(metadata, Value()),\n", + " )\n", + "\n", + " operation = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"gsod-\" + TIMESTAMP, DATA_SCHEMA, src_uri=IMPORT_FILE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML tabular regression model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_pipeline:automl" + }, + "source": [ + "### Create a training pipeline\n", + "\n", + "You may ask, what do we use a pipeline for? You typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `pipeline_name`: A human readable name for the pipeline job.\n", + "- `model_name`: A human readable name for the model.\n", + "- `dataset`: The Vertex fully qualified dataset identifier.\n", + "- `schema`: The dataset labeling (annotation) training schema.\n", + "- `task`: A dictionary describing the requirements for the training job.\n", + "\n", + "The helper function calls the `Pipeline` client service'smethod `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "Let's look now deeper into the *minimal* requirements for constructing a `training_pipeline` specification:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The dataset labeling (annotation) training schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A human readable name for the model.\n", + "- `input_data_config`: The dataset specification.\n", + " - `dataset_id`: The Vertex dataset identifier only (non-fully qualified) -- this is the last part of the fully-qualified identifier.\n", + " - `fraction_split`: If specified, the percentages of the dataset to use for training, test and validation. Otherwise, the percentages are automatically selected by AutoML." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_pipeline:automl" + }, + "outputs": [], + "source": [ + "def create_pipeline(pipeline_name, model_name, dataset, schema, task):\n", + "\n", + " dataset_id = dataset.split(\"/\")[-1]\n", + "\n", + " input_config = {\n", + " \"dataset_id\": dataset_id,\n", + " \"fraction_split\": {\n", + " \"training_fraction\": 0.8,\n", + " \"validation_fraction\": 0.1,\n", + " \"test_fraction\": 0.1,\n", + " },\n", + " }\n", + "\n", + " training_pipeline = {\n", + " \"display_name\": pipeline_name,\n", + " \"training_task_definition\": schema,\n", + " \"training_task_inputs\": task,\n", + " \"input_data_config\": input_config,\n", + " \"model_to_upload\": {\"display_name\": model_name},\n", + " }\n", + "\n", + " try:\n", + " pipeline = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " )\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "task_requirements:automl,tabular" + }, + "source": [ + "### Construct the task requirements\n", + "\n", + "Next, construct the task requirements. Unlike other parameters which take a Python (JSON-like) dictionary, the `task` field takes a Google protobuf Struct, which is very similar to a Python dictionary. Use the `json_format.ParseDict` method for the conversion.\n", + "\n", + "The minimal fields you need to specify are:\n", + "\n", + "- `prediction_type`: Whether we are doing \"classification\" or \"regression\".\n", + "- `target_column`: The CSV heading column name for the column we want to predict (i.e., the label).\n", + "- `train_budget_milli_node_hours`: The maximum time to budget (billed) for training the model, where 1000 = 1 hour.\n", + "- `disable_early_stopping`: Whether True/False to let AutoML use its judgement to stop training early or train for the entire budget.\n", + "- `transformations`: Specifies the feature engineering for each feature column.\n", + "\n", + "For `transformations`, the list must have an entry for each column. The outer key field indicates the type of feature engineering for the corresponding column. In this tutorial, you set it to `\"auto\"` to tell AutoML to automatically determine it.\n", + "\n", + "Finally, create the pipeline by calling the helper function `create_pipeline`, which returns an instance of a training pipeline object." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_transformations:automl,tabular,gsod" + }, + "outputs": [], + "source": [ + "TRANSFORMATIONS = [\n", + " {\"auto\": {\"column_name\": \"year\"}},\n", + " {\"auto\": {\"column_name\": \"month\"}},\n", + " {\"auto\": {\"column_name\": \"day\"}},\n", + "]\n", + "\n", + "label_column = \"mean_temp\"" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_requirements:automl,tabular,transformations,regression" + }, + "outputs": [], + "source": [ + "PIPE_NAME = \"gsod_pipe-\" + TIMESTAMP\n", + "MODEL_NAME = \"gsod_model-\" + TIMESTAMP\n", + "\n", + "task = Value(\n", + " struct_value=Struct(\n", + " fields={\n", + " \"target_column\": Value(string_value=label_column),\n", + " \"prediction_type\": Value(string_value=\"regression\"),\n", + " \"train_budget_milli_node_hours\": Value(number_value=1000),\n", + " \"disable_early_stopping\": Value(bool_value=False),\n", + " \"transformations\": json_format.ParseDict(TRANSFORMATIONS, Value()),\n", + " }\n", + " )\n", + ")\n", + "\n", + "response = create_pipeline(PIPE_NAME, MODEL_NAME, dataset_id, TRAINING_SCHEMA, task)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 30 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "model_information" + }, + "source": [ + "## Model information\n", + "\n", + "Now that your model is trained, you can get some information on your model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:automl" + }, + "source": [ + "## Evaluate the Model resource\n", + "\n", + "Now find out how good the model service believes your model is. As part of training, some portion of the dataset was set aside as the test (holdout) data, which is used by the pipeline service to evaluate the model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "list_model_evaluations:automl,lrg" + }, + "source": [ + "### List evaluations for all slices\n", + "\n", + "Use this helper function `list_model_evaluations`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified model identifier for the `Model` resource.\n", + "\n", + "This helper function uses the model client service's `list_model_evaluations` method, which takes the same parameter. The response object from the call is a list, where each element is an evaluation metric.\n", + "\n", + "For each evaluation -- you probably only have one, we then print all the key names for each metric in the evaluation, and for a small set (`rootMeanSquaredError` and `meanAbsoluteError`) you will print the result." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "list_model_evaluations:automl,lrg" + }, + "outputs": [], + "source": [ + "def list_model_evaluations(name):\n", + " response = clients[\"model\"].list_model_evaluations(parent=name)\n", + " for evaluation in response:\n", + " print(\"model_evaluation\")\n", + " print(\" name:\", evaluation.name)\n", + " print(\" metrics_schema_uri:\", evaluation.metrics_schema_uri)\n", + " metrics = json_format.MessageToDict(evaluation._pb.metrics)\n", + " for metric in metrics.keys():\n", + " print(metric)\n", + " print(\"rootMeanSquaredError\", metrics[\"rootMeanSquaredError\"])\n", + " print(\"meanAbsoluteError\", metrics[\"meanAbsoluteError\"])\n", + "\n", + " return evaluation.name\n", + "\n", + "\n", + "last_evaluation = list_model_evaluations(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:automl" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created with AutoML. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"gsod_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:dedicated" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `dedicated_resources`: This refers to how many compute instances (replicas) that are scaled for serving prediction requests.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `min_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:dedicated" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"gsod_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " \"machine_spec\": machine_spec,\n", + " },\n", + " \"disable_container_logging\": False,\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_test_item:automl,online_prediction" + }, + "source": [ + "### Make test item\n", + "\n", + "You will use synthetic data as a test data item. Don't be concerned that we are using synthetic data -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_test_item:automl,tabular,gsod" + }, + "outputs": [], + "source": [ + "INSTANCE = {\"year\": \"1932\", \"month\": \"11\", \"day\": \"6\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "predict_item:automl,lrg" + }, + "source": [ + "### Make a prediction\n", + "\n", + "Now you have a test item. Use this helper function `predict_item`, which takes the following parameters:\n", + "\n", + "- `filename`: The Cloud Storage path to the test item.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `parameters_dict`: Additional filtering parameters for serving prediction results.\n", + "\n", + "This function calls the prediction client service's `predict` method with the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `instances`: A list of instances (data items) to predict.\n", + "- `parameters`: Additional filtering parameters for serving prediction results. *Note*, tabular models do not support additional parameters.\n", + "\n", + "#### Request\n", + "\n", + "The format of each instance is, where values must be specified as a string:\n", + "\n", + " { 'feature_1': 'value_1', 'feature_2': 'value_2', ... }\n", + "\n", + "Since the `predict()` method can take multiple items (instances), you send your single test item as a list of one test item. As a final step, you package the instances list into Google's protobuf format -- which is what we pass to the `predict()` method.\n", + "\n", + "#### Response\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding image in the request. You will see in the output for each prediction -- in this case there is just one:\n", + "\n", + "- `value`: The predicted value." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "predict_item:automl,lrg" + }, + "outputs": [], + "source": [ + "def predict_item(data, endpoint, parameters_dict):\n", + " parameters = json_format.ParseDict(parameters_dict, Value())\n", + "\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [data]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", dict(prediction))\n", + "\n", + "\n", + "predict_item(INSTANCE, endpoint_id, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_automl_tabular_regression_online_bq.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/automl/showcase_automl_text_classification_batch.ipynb b/notebooks/community/gapic/automl/showcase_automl_text_classification_batch.ipynb new file mode 100644 index 000000000..ab7badaf5 --- /dev/null +++ b/notebooks/community/gapic/automl/showcase_automl_text_classification_batch.ipynb @@ -0,0 +1,1778 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: AutoML text classification model for batch prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to create text classification models and do batch prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:happydb,tcn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Happy Moments dataset](https://www.kaggle.com/ritresearch/happydb) from [Kaggle Datasets](https://www.kaggle.com/ritresearch/happydb). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,batch_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML text classification model from a Python script, and then do a batch prediction using the Vertex client library. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Make a batch prediction.\n", + "\n", + "There is one key difference between using batch prediction and using online prediction:\n", + "\n", + "* Prediction Service: Does an on-demand prediction for the entire set of instances (i.e., one or more data items) and returns the results in real-time.\n", + "\n", + "* Batch Prediction Service: Does a queued (batch) prediction for the entire set of instances in the background and stores the results in a Cloud Storage bucket when ready." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Set constants unique to AutoML datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the `Dataset` resource service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the `Dataset` resource service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:tcn" + }, + "outputs": [], + "source": [ + "# Text Dataset type\n", + "DATA_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/metadata/text_1.0.0.yaml\"\n", + "# Text Labeling type\n", + "LABEL_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/ioformat/text_classification_single_label_io_format_1.0.0.yaml\"\n", + "# Text Training task\n", + "TRAINING_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_text_classification_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:prediction" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for prediction.\n", + "\n", + "Set the variable `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:automl" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "For AutoML batch prediction, the container image for the serving binary is pre-determined by the Vertex prediction service. More specifically, the service will pick the appropriate container for the model depending on the hardware accelerator you selected." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for prediction.\n", + "\n", + "- Set the variable `DEPLOY_COMPUTE` to configure the compute resources for the VM you will use for prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML text classification model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:automl,batch_prediction" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training.\n", + "- Job Service for batch prediction and custom training." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:automl,batch_prediction" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"job\"] = create_job_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "\n", + "def create_dataset(name, schema, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " dataset = aip.Dataset(\n", + " display_name=name, metadata_schema_uri=schema, labels=labels\n", + " )\n", + "\n", + " operation = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"happydb-\" + TIMESTAMP, DATA_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:text,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for text has a couple of requirements for your text data.\n", + "\n", + "- Text examples must be stored in a CSV or JSONL file." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:tcn,u_dataset,csv" + }, + "source": [ + "#### CSV\n", + "\n", + "For text classification, the CSV file has a few requirements:\n", + "\n", + "- No heading.\n", + "- First column is the text example or Cloud Storage path to text file (.txt suffix).\n", + "- Second column the label." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:happydb,csv,tcn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://cloud-ml-data/NL-classification/happiness.csv\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:csv" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Happy Moments dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:csv" + }, + "outputs": [], + "source": [ + "if \"IMPORT_FILES\" in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_data" + }, + "source": [ + "### Import data\n", + "\n", + "Now, import the data into your Vertex Dataset resource. Use this helper function `import_data` to import the data. The function does the following:\n", + "\n", + "- Uses the `Dataset` client.\n", + "- Calls the client method `import_data`, with the following parameters:\n", + " - `name`: The human readable name you give to the `Dataset` resource (e.g., happydb).\n", + " - `import_configs`: The import configuration.\n", + "\n", + "- `import_configs`: A Python list containing a dictionary, with the key/value entries:\n", + " - `gcs_sources`: A list of URIs to the paths of the one or more index files.\n", + " - `import_schema_uri`: The schema identifying the labeling type.\n", + "\n", + "The `import_data()` method returns a long running `operation` object. This will take a few minutes to complete. If you are in a live tutorial, this would be a good time to ask questions, or take a personal break." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_data" + }, + "outputs": [], + "source": [ + "def import_data(dataset, gcs_sources, schema):\n", + " config = [{\"gcs_source\": {\"uris\": gcs_sources}, \"import_schema_uri\": schema}]\n", + " print(\"dataset:\", dataset_id)\n", + " start_time = time.time()\n", + " try:\n", + " operation = clients[\"dataset\"].import_data(\n", + " name=dataset_id, import_configs=config\n", + " )\n", + " print(\"Long running operation:\", operation.operation.name)\n", + "\n", + " result = operation.result()\n", + " print(\"result:\", result)\n", + " print(\"time:\", int(time.time() - start_time), \"secs\")\n", + " print(\"error:\", operation.exception())\n", + " print(\"meta :\", operation.metadata)\n", + " print(\n", + " \"after: running:\",\n", + " operation.running(),\n", + " \"done:\",\n", + " operation.done(),\n", + " \"cancelled:\",\n", + " operation.cancelled(),\n", + " )\n", + "\n", + " return operation\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "import_data(dataset_id, [IMPORT_FILE], LABEL_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML text classification model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_pipeline:automl" + }, + "source": [ + "### Create a training pipeline\n", + "\n", + "You may ask, what do we use a pipeline for? You typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `pipeline_name`: A human readable name for the pipeline job.\n", + "- `model_name`: A human readable name for the model.\n", + "- `dataset`: The Vertex fully qualified dataset identifier.\n", + "- `schema`: The dataset labeling (annotation) training schema.\n", + "- `task`: A dictionary describing the requirements for the training job.\n", + "\n", + "The helper function calls the `Pipeline` client service'smethod `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "Let's look now deeper into the *minimal* requirements for constructing a `training_pipeline` specification:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The dataset labeling (annotation) training schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A human readable name for the model.\n", + "- `input_data_config`: The dataset specification.\n", + " - `dataset_id`: The Vertex dataset identifier only (non-fully qualified) -- this is the last part of the fully-qualified identifier.\n", + " - `fraction_split`: If specified, the percentages of the dataset to use for training, test and validation. Otherwise, the percentages are automatically selected by AutoML." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_pipeline:automl" + }, + "outputs": [], + "source": [ + "def create_pipeline(pipeline_name, model_name, dataset, schema, task):\n", + "\n", + " dataset_id = dataset.split(\"/\")[-1]\n", + "\n", + " input_config = {\n", + " \"dataset_id\": dataset_id,\n", + " \"fraction_split\": {\n", + " \"training_fraction\": 0.8,\n", + " \"validation_fraction\": 0.1,\n", + " \"test_fraction\": 0.1,\n", + " },\n", + " }\n", + "\n", + " training_pipeline = {\n", + " \"display_name\": pipeline_name,\n", + " \"training_task_definition\": schema,\n", + " \"training_task_inputs\": task,\n", + " \"input_data_config\": input_config,\n", + " \"model_to_upload\": {\"display_name\": model_name},\n", + " }\n", + "\n", + " try:\n", + " pipeline = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " )\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "task_requirements:automl,tcn" + }, + "source": [ + "### Construct the task requirements\n", + "\n", + "Next, construct the task requirements. Unlike other parameters which take a Python (JSON-like) dictionary, the `task` field takes a Google protobuf Struct, which is very similar to a Python dictionary. Use the `json_format.ParseDict` method for the conversion.\n", + "\n", + "The minimal fields you need to specify are:\n", + "\n", + "- `multi_label`: Whether True/False this is a multi-label (vs single) classification.\n", + "\n", + "Finally, create the pipeline by calling the helper function `create_pipeline`, which returns an instance of a training pipeline object." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_requirements:automl,tcn" + }, + "outputs": [], + "source": [ + "PIPE_NAME = \"happydb_pipe-\" + TIMESTAMP\n", + "MODEL_NAME = \"happydb_model-\" + TIMESTAMP\n", + "\n", + "task = json_format.ParseDict(\n", + " {\n", + " \"multi_label\": False,\n", + " },\n", + " Value(),\n", + ")\n", + "\n", + "response = create_pipeline(PIPE_NAME, MODEL_NAME, dataset_id, TRAINING_SCHEMA, task)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 240 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "model_information" + }, + "source": [ + "## Model information\n", + "\n", + "Now that your model is trained, you can get some information on your model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:automl" + }, + "source": [ + "## Evaluate the Model resource\n", + "\n", + "Now find out how good the model service believes your model is. As part of training, some portion of the dataset was set aside as the test (holdout) data, which is used by the pipeline service to evaluate the model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "list_model_evaluations:automl,tcn" + }, + "source": [ + "### List evaluations for all slices\n", + "\n", + "Use this helper function `list_model_evaluations`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified model identifier for the `Model` resource.\n", + "\n", + "This helper function uses the model client service's `list_model_evaluations` method, which takes the same parameter. The response object from the call is a list, where each element is an evaluation metric.\n", + "\n", + "For each evaluation (you probably only have one) we then print all the key names for each metric in the evaluation, and for a small set (`logLoss` and `auPrc`) you will print the result." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "list_model_evaluations:automl,tcn" + }, + "outputs": [], + "source": [ + "def list_model_evaluations(name):\n", + " response = clients[\"model\"].list_model_evaluations(parent=name)\n", + " for evaluation in response:\n", + " print(\"model_evaluation\")\n", + " print(\" name:\", evaluation.name)\n", + " print(\" metrics_schema_uri:\", evaluation.metrics_schema_uri)\n", + " metrics = json_format.MessageToDict(evaluation._pb.metrics)\n", + " for metric in metrics.keys():\n", + " print(metric)\n", + " print(\"logloss\", metrics[\"logLoss\"])\n", + " print(\"auPrc\", metrics[\"auPrc\"])\n", + "\n", + " return evaluation.name\n", + "\n", + "\n", + "last_evaluation = list_model_evaluations(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy:batch_prediction" + }, + "source": [ + "## Model deployment for batch prediction\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created for batch prediction. This differs from deploying a `Model` resource for on-demand prediction.\n", + "\n", + "For online prediction, you:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource.\n", + "\n", + "3. Make online prediction requests to the `Endpoint` resource.\n", + "\n", + "For batch-prediction, you:\n", + "\n", + "1. Create a batch prediction job.\n", + "\n", + "2. The job service will provision resources for the batch prediction request.\n", + "\n", + "3. The results of the batch prediction request are returned to the caller.\n", + "\n", + "4. The job service will unprovision the resoures for the batch prediction request." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a batch prediction request\n", + "\n", + "Now do a batch prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_items:batch_prediction" + }, + "source": [ + "### Get test item(s)\n", + "\n", + "Now do a batch prediction to your Vertex model. You will use arbitrary examples out of the dataset as a test items. Don't be concerned that the examples were likely used in training the model -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_items:automl,tcn,csv" + }, + "outputs": [], + "source": [ + "test_items = ! gsutil cat $IMPORT_FILE | head -n2\n", + "if len(test_items[0]) == 3:\n", + " _, test_item_1, test_label_1 = str(test_items[0]).split(\",\")\n", + " _, test_item_2, test_label_2 = str(test_items[1]).split(\",\")\n", + "else:\n", + " test_item_1, test_label_1 = str(test_items[0]).split(\",\")\n", + " test_item_2, test_label_2 = str(test_items[1]).split(\",\")\n", + "\n", + "print(test_item_1, test_label_1)\n", + "print(test_item_2, test_label_2)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:automl,text" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "Now make a batch input file, which you will store in your local Cloud Storage bucket. The batch input file can only be in JSONL format. For JSONL file, you make one dictionary entry per line for each data item (instance). The dictionary contains the key/value pairs:\n", + "\n", + "- `content`: The Cloud Storage path to the file with the text item.\n", + "- `mime_type`: The content type. In our example, it is an `text` file.\n", + "\n", + "For example:\n", + "\n", + " {'content': '[your-bucket]/file1.txt', 'mime_type': 'text'}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_file:automl,text" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "import tensorflow as tf\n", + "\n", + "gcs_test_item_1 = BUCKET_NAME + \"/test1.txt\"\n", + "with tf.io.gfile.GFile(gcs_test_item_1, \"w\") as f:\n", + " f.write(test_item_1 + \"\\n\")\n", + "gcs_test_item_2 = BUCKET_NAME + \"/test2.txt\"\n", + "with tf.io.gfile.GFile(gcs_test_item_2, \"w\") as f:\n", + " f.write(test_item_2 + \"\\n\")\n", + "\n", + "gcs_input_uri = BUCKET_NAME + \"/test.jsonl\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " data = {\"content\": gcs_test_item_1, \"mime_type\": \"text/plain\"}\n", + " f.write(json.dumps(data) + \"\\n\")\n", + " data = {\"content\": gcs_test_item_2, \"mime_type\": \"text/plain\"}\n", + " f.write(json.dumps(data) + \"\\n\")\n", + "\n", + "print(gcs_input_uri)\n", + "! gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your batch prediction requests:\n", + "\n", + "- Single Instance: The batch prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The batch prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and batch prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The batch prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_request:automl,tcn" + }, + "source": [ + "### Make batch prediction request\n", + "\n", + "Now that your batch of two test items is ready, let's do the batch request. Use this helper function `create_batch_prediction_job`, with the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the prediction job.\n", + "- `model_name`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `gcs_source_uri`: The Cloud Storage path to the input file -- which you created above.\n", + "- `gcs_destination_output_uri_prefix`: The Cloud Storage path that the service will write the predictions to.\n", + "- `parameters`: Additional filtering parameters for serving prediction results.\n", + "\n", + "The helper function calls the job client service's `create_batch_prediction_job` metho, with the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for Dataset, Model and Pipeline resources.\n", + "- `batch_prediction_job`: The specification for the batch prediction job.\n", + "\n", + "Let's now dive into the specification for the `batch_prediction_job`:\n", + "\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `dedicated_resources`: The compute resources to provision for the batch prediction job.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `starting_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "- `model_parameters`: Additional filtering parameters for serving prediction results. *Note*, text models do not support additional parameters.\n", + "- `input_config`: The input source and format type for the instances to predict.\n", + " - `instances_format`: The format of the batch prediction request file: `jsonl` only supported.\n", + " - `gcs_source`: A list of one or more Cloud Storage paths to your batch prediction requests.\n", + "- `output_config`: The output destination and format for the predictions.\n", + " - `prediction_format`: The format of the batch prediction response file: `jsonl` only supported.\n", + " - `gcs_destination`: The output destination for the predictions.\n", + "- `dedicated_resources`: The compute resources to provision for the batch prediction job.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `starting_replica_count`: The number of compute instances to initially provision.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to. In this tutorial, only one instance is provisioned.\n", + "\n", + "This call is an asychronous operation. You will print from the response object a few select fields, including:\n", + "\n", + "- `name`: The Vertex fully qualified identifier assigned to the batch prediction job.\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the Model resource.\n", + "- `generate_explanations`: Whether True/False explanations were provided with the predictions (explainability).\n", + "- `state`: The state of the prediction job (pending, running, etc).\n", + "\n", + "Since this call will take a few moments to execute, you will likely get `JobState.JOB_STATE_PENDING` for `state`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_request:automl,tcn" + }, + "outputs": [], + "source": [ + "BATCH_MODEL = \"happydb_batch-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_batch_prediction_job(\n", + " display_name,\n", + " model_name,\n", + " gcs_source_uri,\n", + " gcs_destination_output_uri_prefix,\n", + " parameters=None,\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " batch_prediction_job = {\n", + " \"display_name\": display_name,\n", + " # Format: 'projects/{project}/locations/{location}/models/{model_id}'\n", + " \"model\": model_name,\n", + " \"model_parameters\": json_format.ParseDict(parameters, Value()),\n", + " \"input_config\": {\n", + " \"instances_format\": IN_FORMAT,\n", + " \"gcs_source\": {\"uris\": [gcs_source_uri]},\n", + " },\n", + " \"output_config\": {\n", + " \"predictions_format\": OUT_FORMAT,\n", + " \"gcs_destination\": {\"output_uri_prefix\": gcs_destination_output_uri_prefix},\n", + " },\n", + " \"dedicated_resources\": {\n", + " \"machine_spec\": machine_spec,\n", + " \"starting_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " },\n", + " }\n", + " response = clients[\"job\"].create_batch_prediction_job(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + " )\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try:\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", response.labels)\n", + " return response\n", + "\n", + "\n", + "IN_FORMAT = \"jsonl\"\n", + "OUT_FORMAT = \"jsonl\" # [jsonl]\n", + "\n", + "response = create_batch_prediction_job(\n", + " BATCH_MODEL, model_to_deploy_id, gcs_input_uri, BUCKET_NAME, None\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batch_job_id:response" + }, + "source": [ + "Now get the unique identifier for the batch prediction job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batch_job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the batch job\n", + "batch_job_id = response.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split(\"/\")[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_batch_prediction_job" + }, + "source": [ + "### Get information on a batch prediction job\n", + "\n", + "Use this helper function `get_batch_prediction_job`, with the following paramter:\n", + "\n", + "- `job_name`: The Vertex fully qualified identifier for the batch prediction job.\n", + "\n", + "The helper function calls the job client service's `get_batch_prediction_job` method, with the following paramter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the batch prediction job. In this tutorial, you will pass it the Vertex fully qualified identifier for your batch prediction job -- `batch_job_id`\n", + "\n", + "The helper function will return the Cloud Storage path to where the predictions are stored -- `gcs_destination`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_batch_prediction_job" + }, + "outputs": [], + "source": [ + "def get_batch_prediction_job(job_name, silent=False):\n", + " response = clients[\"job\"].get_batch_prediction_job(name=job_name)\n", + " if silent:\n", + " return response.output_config.gcs_destination.output_uri_prefix, response.state\n", + "\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try: # not all data types support explanations\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" error:\", response.error)\n", + " gcs_destination = response.output_config.gcs_destination\n", + " print(\" gcs_destination\")\n", + " print(\" output_uri_prefix:\", gcs_destination.output_uri_prefix)\n", + " return gcs_destination.output_uri_prefix, response.state\n", + "\n", + "\n", + "predictions, state = get_batch_prediction_job(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_the_predictions:automl,tcn" + }, + "source": [ + "### Get the predictions\n", + "\n", + "When the batch prediction is done processing, the job state will be `JOB_STATE_SUCCEEDED`.\n", + "\n", + "Finally you view the predictions stored at the Cloud Storage path you set as output. The predictions will be in a JSONL format, which you indicated at the time you made the batch prediction job, under a subfolder starting with the name `prediction`, and under that folder will be a file called `predictions*.jsonl`.\n", + "\n", + "Now display (cat) the contents. You will see multiple JSON objects, one for each prediction.\n", + "\n", + "The first field `text_snippet` is the text file you did the prediction on, and the second field `annotations` is the prediction, which is further broken down into:\n", + "\n", + "- `score`: The percent of confidence between 0 and 1.\n", + "- `display_name`: The corresponding class name." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_the_predictions:automl,text" + }, + "outputs": [], + "source": [ + "def get_latest_predictions(gcs_out_dir):\n", + " \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n", + " folders = !gsutil ls $gcs_out_dir\n", + " latest = \"\"\n", + " for folder in folders:\n", + " subfolder = folder.split(\"/\")[-2]\n", + " if subfolder.startswith(\"prediction-\"):\n", + " if subfolder > latest:\n", + " latest = folder[:-1]\n", + " return latest\n", + "\n", + "\n", + "while True:\n", + " predictions, state = get_batch_prediction_job(batch_job_id, True)\n", + " if state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", state)\n", + " if state == aip.JobState.JOB_STATE_FAILED:\n", + " raise Exception(\"Batch Job Failed\")\n", + " else:\n", + " folder = get_latest_predictions(predictions)\n", + " ! gsutil ls $folder/prediction*.jsonl\n", + "\n", + " ! gsutil cat $folder/prediction*.jsonl\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_automl_text_classification_batch.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/automl/showcase_automl_text_classification_online.ipynb b/notebooks/community/gapic/automl/showcase_automl_text_classification_online.ipynb new file mode 100644 index 000000000..328f23bba --- /dev/null +++ b/notebooks/community/gapic/automl/showcase_automl_text_classification_online.ipynb @@ -0,0 +1,1552 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: AutoML text classification model for online prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to create text classification models and do online prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:happydb,tcn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Happy Moments dataset](https://www.kaggle.com/ritresearch/happydb) from [Kaggle Datasets](https://www.kaggle.com/ritresearch/happydb). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,online_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML text classification model and deploy for online prediction from a Python script using the Vertex client library. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model`." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Set constants unique to AutoML datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the `Dataset` resource service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the `Dataset` resource service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:tcn" + }, + "outputs": [], + "source": [ + "# Text Dataset type\n", + "DATA_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/metadata/text_1.0.0.yaml\"\n", + "# Text Labeling type\n", + "LABEL_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/ioformat/text_classification_single_label_io_format_1.0.0.yaml\"\n", + "# Text Training task\n", + "TRAINING_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_text_classification_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML text classification model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:automl,online_prediction" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training.\n", + "- Endpoint Service for deployment.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:automl,online_prediction" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "\n", + "def create_dataset(name, schema, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " dataset = aip.Dataset(\n", + " display_name=name, metadata_schema_uri=schema, labels=labels\n", + " )\n", + "\n", + " operation = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"happydb-\" + TIMESTAMP, DATA_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:text,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for text has a couple of requirements for your text data.\n", + "\n", + "- Text examples must be stored in a CSV or JSONL file." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:tcn,u_dataset,csv" + }, + "source": [ + "#### CSV\n", + "\n", + "For text classification, the CSV file has a few requirements:\n", + "\n", + "- No heading.\n", + "- First column is the text example or Cloud Storage path to text file (.txt suffix).\n", + "- Second column the label." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:happydb,csv,tcn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://cloud-ml-data/NL-classification/happiness.csv\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:csv" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Happy Moments dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:csv" + }, + "outputs": [], + "source": [ + "if \"IMPORT_FILES\" in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_data" + }, + "source": [ + "### Import data\n", + "\n", + "Now, import the data into your Vertex Dataset resource. Use this helper function `import_data` to import the data. The function does the following:\n", + "\n", + "- Uses the `Dataset` client.\n", + "- Calls the client method `import_data`, with the following parameters:\n", + " - `name`: The human readable name you give to the `Dataset` resource (e.g., happydb).\n", + " - `import_configs`: The import configuration.\n", + "\n", + "- `import_configs`: A Python list containing a dictionary, with the key/value entries:\n", + " - `gcs_sources`: A list of URIs to the paths of the one or more index files.\n", + " - `import_schema_uri`: The schema identifying the labeling type.\n", + "\n", + "The `import_data()` method returns a long running `operation` object. This will take a few minutes to complete. If you are in a live tutorial, this would be a good time to ask questions, or take a personal break." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_data" + }, + "outputs": [], + "source": [ + "def import_data(dataset, gcs_sources, schema):\n", + " config = [{\"gcs_source\": {\"uris\": gcs_sources}, \"import_schema_uri\": schema}]\n", + " print(\"dataset:\", dataset_id)\n", + " start_time = time.time()\n", + " try:\n", + " operation = clients[\"dataset\"].import_data(\n", + " name=dataset_id, import_configs=config\n", + " )\n", + " print(\"Long running operation:\", operation.operation.name)\n", + "\n", + " result = operation.result()\n", + " print(\"result:\", result)\n", + " print(\"time:\", int(time.time() - start_time), \"secs\")\n", + " print(\"error:\", operation.exception())\n", + " print(\"meta :\", operation.metadata)\n", + " print(\n", + " \"after: running:\",\n", + " operation.running(),\n", + " \"done:\",\n", + " operation.done(),\n", + " \"cancelled:\",\n", + " operation.cancelled(),\n", + " )\n", + "\n", + " return operation\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "import_data(dataset_id, [IMPORT_FILE], LABEL_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML text classification model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_pipeline:automl" + }, + "source": [ + "### Create a training pipeline\n", + "\n", + "You may ask, what do we use a pipeline for? You typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `pipeline_name`: A human readable name for the pipeline job.\n", + "- `model_name`: A human readable name for the model.\n", + "- `dataset`: The Vertex fully qualified dataset identifier.\n", + "- `schema`: The dataset labeling (annotation) training schema.\n", + "- `task`: A dictionary describing the requirements for the training job.\n", + "\n", + "The helper function calls the `Pipeline` client service'smethod `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "Let's look now deeper into the *minimal* requirements for constructing a `training_pipeline` specification:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The dataset labeling (annotation) training schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A human readable name for the model.\n", + "- `input_data_config`: The dataset specification.\n", + " - `dataset_id`: The Vertex dataset identifier only (non-fully qualified) -- this is the last part of the fully-qualified identifier.\n", + " - `fraction_split`: If specified, the percentages of the dataset to use for training, test and validation. Otherwise, the percentages are automatically selected by AutoML." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_pipeline:automl" + }, + "outputs": [], + "source": [ + "def create_pipeline(pipeline_name, model_name, dataset, schema, task):\n", + "\n", + " dataset_id = dataset.split(\"/\")[-1]\n", + "\n", + " input_config = {\n", + " \"dataset_id\": dataset_id,\n", + " \"fraction_split\": {\n", + " \"training_fraction\": 0.8,\n", + " \"validation_fraction\": 0.1,\n", + " \"test_fraction\": 0.1,\n", + " },\n", + " }\n", + "\n", + " training_pipeline = {\n", + " \"display_name\": pipeline_name,\n", + " \"training_task_definition\": schema,\n", + " \"training_task_inputs\": task,\n", + " \"input_data_config\": input_config,\n", + " \"model_to_upload\": {\"display_name\": model_name},\n", + " }\n", + "\n", + " try:\n", + " pipeline = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " )\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "task_requirements:automl,tcn" + }, + "source": [ + "### Construct the task requirements\n", + "\n", + "Next, construct the task requirements. Unlike other parameters which take a Python (JSON-like) dictionary, the `task` field takes a Google protobuf Struct, which is very similar to a Python dictionary. Use the `json_format.ParseDict` method for the conversion.\n", + "\n", + "The minimal fields you need to specify are:\n", + "\n", + "- `multi_label`: Whether True/False this is a multi-label (vs single) classification.\n", + "\n", + "Finally, create the pipeline by calling the helper function `create_pipeline`, which returns an instance of a training pipeline object." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_requirements:automl,tcn" + }, + "outputs": [], + "source": [ + "PIPE_NAME = \"happydb_pipe-\" + TIMESTAMP\n", + "MODEL_NAME = \"happydb_model-\" + TIMESTAMP\n", + "\n", + "task = json_format.ParseDict(\n", + " {\n", + " \"multi_label\": False,\n", + " },\n", + " Value(),\n", + ")\n", + "\n", + "response = create_pipeline(PIPE_NAME, MODEL_NAME, dataset_id, TRAINING_SCHEMA, task)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 120 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "model_information" + }, + "source": [ + "## Model information\n", + "\n", + "Now that your model is trained, you can get some information on your model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:automl" + }, + "source": [ + "## Evaluate the Model resource\n", + "\n", + "Now find out how good the model service believes your model is. As part of training, some portion of the dataset was set aside as the test (holdout) data, which is used by the pipeline service to evaluate the model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "list_model_evaluations:automl,tcn" + }, + "source": [ + "### List evaluations for all slices\n", + "\n", + "Use this helper function `list_model_evaluations`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified model identifier for the `Model` resource.\n", + "\n", + "This helper function uses the model client service's `list_model_evaluations` method, which takes the same parameter. The response object from the call is a list, where each element is an evaluation metric.\n", + "\n", + "For each evaluation (you probably only have one) we then print all the key names for each metric in the evaluation, and for a small set (`logLoss` and `auPrc`) you will print the result." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "list_model_evaluations:automl,tcn" + }, + "outputs": [], + "source": [ + "def list_model_evaluations(name):\n", + " response = clients[\"model\"].list_model_evaluations(parent=name)\n", + " for evaluation in response:\n", + " print(\"model_evaluation\")\n", + " print(\" name:\", evaluation.name)\n", + " print(\" metrics_schema_uri:\", evaluation.metrics_schema_uri)\n", + " metrics = json_format.MessageToDict(evaluation._pb.metrics)\n", + " for metric in metrics.keys():\n", + " print(metric)\n", + " print(\"logloss\", metrics[\"logLoss\"])\n", + " print(\"auPrc\", metrics[\"auPrc\"])\n", + "\n", + " return evaluation.name\n", + "\n", + "\n", + "last_evaluation = list_model_evaluations(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:automl" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created with AutoML. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"happydb_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:automatic" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `automatic_resources`: This refers to how many redundant compute instances (replicas). For this example, we set it to one (no replication).\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:automatic" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"happydb_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"automatic_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " },\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_item" + }, + "source": [ + "### Get test item\n", + "\n", + "You will use an arbitrary example out of the dataset as a test item. Don't be concerned that the example was likely used in training the model -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:automl,tcn,csv" + }, + "outputs": [], + "source": [ + "test_item = ! gsutil cat $IMPORT_FILE | head -n1\n", + "if len(test_item[0]) == 3:\n", + " _, test_item, test_label = str(test_item[0]).split(\",\")\n", + "else:\n", + " test_item, test_label = str(test_item[0]).split(\",\")\n", + "\n", + "print(test_item, test_label)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "predict_item:automl,tcn" + }, + "source": [ + "### Make a prediction\n", + "\n", + "Now you have a test item. Use this helper function `predict_item`, which takes the following parameters:\n", + "\n", + "- `filename`: The Cloud Storage path to the test item.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `parameters_dict`: Additional filtering parameters for serving prediction results.\n", + "\n", + "This function calls the prediction client service's `predict` method with the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `instances`: A list of instances (text files) to predict.\n", + "- `parameters`: Additional filtering parameters for serving prediction results. *Note*, text models do not support additional parameters.\n", + "\n", + "#### Request\n", + "\n", + "The format of each instance is:\n", + "\n", + " { 'content': text_item }\n", + "\n", + "Since the `predict()` method can take multiple items (instances), you send your single test item as a list of one test item. As a final step, you package the instances list into Google's protobuf format -- which is what you pass to the `predict()` method.\n", + "\n", + "#### Response\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding text in the request. You will see in the output for each prediction -- in this case there is just one:\n", + "\n", + "- `confidences`: Confidence level in the prediction.\n", + "- `displayNames`: The predicted label." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "predict_item:automl,tcn" + }, + "outputs": [], + "source": [ + "def predict_item(data, endpoint, parameters_dict):\n", + "\n", + " parameters = json_format.ParseDict(parameters_dict, Value())\n", + "\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [{\"content\": data}]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", dict(prediction))\n", + " return response\n", + "\n", + "\n", + "response = predict_item(test_item, endpoint_id, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_automl_text_classification_online.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/automl/showcase_automl_text_entity_extraction_batch.ipynb b/notebooks/community/gapic/automl/showcase_automl_text_entity_extraction_batch.ipynb new file mode 100644 index 000000000..16f02d2f1 --- /dev/null +++ b/notebooks/community/gapic/automl/showcase_automl_text_entity_extraction_batch.ipynb @@ -0,0 +1,1787 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: AutoML text entity extraction model for batch prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to create text entity extraction models and do batch prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:biomedical,ten" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [NCBI Disease Research Abstracts dataset](https://www.ncbi.nlm.nih.gov/CBBresearch/Dogan/DISEASE/) from [National Center for Biotechnology Information](https://www.ncbi.nlm.nih.gov/). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,batch_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML text entity extraction model from a Python script, and then do a batch prediction using the Vertex client library. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Make a batch prediction.\n", + "\n", + "There is one key difference between using batch prediction and using online prediction:\n", + "\n", + "* Prediction Service: Does an on-demand prediction for the entire set of instances (i.e., one or more data items) and returns the results in real-time.\n", + "\n", + "* Batch Prediction Service: Does a queued (batch) prediction for the entire set of instances in the background and stores the results in a Cloud Storage bucket when ready." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Set constants unique to AutoML datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the `Dataset` resource service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the `Dataset` resource service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:ten" + }, + "outputs": [], + "source": [ + "# Text Dataset type\n", + "DATA_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/metadata/text_1.0.0.yaml\"\n", + "# Text Labeling type\n", + "LABEL_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/ioformat/text_extraction_io_format_1.0.0.yaml\"\n", + "# Text Training task\n", + "TRAINING_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_text_extraction_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:prediction" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for prediction.\n", + "\n", + "Set the variable `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:automl" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "For AutoML batch prediction, the container image for the serving binary is pre-determined by the Vertex prediction service. More specifically, the service will pick the appropriate container for the model depending on the hardware accelerator you selected." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for prediction.\n", + "\n", + "- Set the variable `DEPLOY_COMPUTE` to configure the compute resources for the VM you will use for prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML text entity extraction model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:automl,batch_prediction" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training.\n", + "- Job Service for batch prediction and custom training." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:automl,batch_prediction" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"job\"] = create_job_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "\n", + "def create_dataset(name, schema, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " dataset = aip.Dataset(\n", + " display_name=name, metadata_schema_uri=schema, labels=labels\n", + " )\n", + "\n", + " operation = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"biomedical-\" + TIMESTAMP, DATA_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:text,ten,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for text has a couple of requirements for your text entity extraction data.\n", + "\n", + "- Text examples must be stored in a JSONL file. Unlike text classification and sentiment analysis, a CSV index file is not supported.\n", + "- The examples must be either inline text or reference text files that are in Cloud Storage buckets." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:ten,u_dataset,jsonl" + }, + "source": [ + "#### JSONL\n", + "\n", + "For text entity extraction, the JSONL file has a few requirements:\n", + "\n", + "- Each data item is a separate JSON object, on a separate line.\n", + "- The key/value pair `text_segment_annotations` is a list of character start/end positions in the text per entity with the corresponding label.\n", + " - `display_name`: The label.\n", + " - `start_offset/end_offset`: The character offsets of the start/end of the entity.\n", + "- The key/value pair `text_content` is the text.\n", + "\n", + " {'text_segment_annotations': [{'end_offset': value, 'start_offset': value, 'display_name': label}, ...], 'text_content': text}\n", + "\n", + "*Note*: The dictionary key fields may alternatively be in camelCase. For example, 'display_name' can also be 'displayName'." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,jsonl" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the JSONL index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:biomedical,jsonl,ten" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://ucaip-test-us-central1/dataset/ucaip_ten_dataset.jsonl\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:jsonl" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the NCBI Biomedical dataset that is stored in a public Cloud Storage bucket, using a JSONL index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of objects in a JSONL index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:jsonl" + }, + "outputs": [], + "source": [ + "if \"IMPORT_FILES\" in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_data" + }, + "source": [ + "### Import data\n", + "\n", + "Now, import the data into your Vertex Dataset resource. Use this helper function `import_data` to import the data. The function does the following:\n", + "\n", + "- Uses the `Dataset` client.\n", + "- Calls the client method `import_data`, with the following parameters:\n", + " - `name`: The human readable name you give to the `Dataset` resource (e.g., biomedical).\n", + " - `import_configs`: The import configuration.\n", + "\n", + "- `import_configs`: A Python list containing a dictionary, with the key/value entries:\n", + " - `gcs_sources`: A list of URIs to the paths of the one or more index files.\n", + " - `import_schema_uri`: The schema identifying the labeling type.\n", + "\n", + "The `import_data()` method returns a long running `operation` object. This will take a few minutes to complete. If you are in a live tutorial, this would be a good time to ask questions, or take a personal break." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_data" + }, + "outputs": [], + "source": [ + "def import_data(dataset, gcs_sources, schema):\n", + " config = [{\"gcs_source\": {\"uris\": gcs_sources}, \"import_schema_uri\": schema}]\n", + " print(\"dataset:\", dataset_id)\n", + " start_time = time.time()\n", + " try:\n", + " operation = clients[\"dataset\"].import_data(\n", + " name=dataset_id, import_configs=config\n", + " )\n", + " print(\"Long running operation:\", operation.operation.name)\n", + "\n", + " result = operation.result()\n", + " print(\"result:\", result)\n", + " print(\"time:\", int(time.time() - start_time), \"secs\")\n", + " print(\"error:\", operation.exception())\n", + " print(\"meta :\", operation.metadata)\n", + " print(\n", + " \"after: running:\",\n", + " operation.running(),\n", + " \"done:\",\n", + " operation.done(),\n", + " \"cancelled:\",\n", + " operation.cancelled(),\n", + " )\n", + "\n", + " return operation\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "import_data(dataset_id, [IMPORT_FILE], LABEL_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML text entity extraction model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_pipeline:automl" + }, + "source": [ + "### Create a training pipeline\n", + "\n", + "You may ask, what do we use a pipeline for? You typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `pipeline_name`: A human readable name for the pipeline job.\n", + "- `model_name`: A human readable name for the model.\n", + "- `dataset`: The Vertex fully qualified dataset identifier.\n", + "- `schema`: The dataset labeling (annotation) training schema.\n", + "- `task`: A dictionary describing the requirements for the training job.\n", + "\n", + "The helper function calls the `Pipeline` client service'smethod `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "Let's look now deeper into the *minimal* requirements for constructing a `training_pipeline` specification:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The dataset labeling (annotation) training schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A human readable name for the model.\n", + "- `input_data_config`: The dataset specification.\n", + " - `dataset_id`: The Vertex dataset identifier only (non-fully qualified) -- this is the last part of the fully-qualified identifier.\n", + " - `fraction_split`: If specified, the percentages of the dataset to use for training, test and validation. Otherwise, the percentages are automatically selected by AutoML." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_pipeline:automl" + }, + "outputs": [], + "source": [ + "def create_pipeline(pipeline_name, model_name, dataset, schema, task):\n", + "\n", + " dataset_id = dataset.split(\"/\")[-1]\n", + "\n", + " input_config = {\n", + " \"dataset_id\": dataset_id,\n", + " \"fraction_split\": {\n", + " \"training_fraction\": 0.8,\n", + " \"validation_fraction\": 0.1,\n", + " \"test_fraction\": 0.1,\n", + " },\n", + " }\n", + "\n", + " training_pipeline = {\n", + " \"display_name\": pipeline_name,\n", + " \"training_task_definition\": schema,\n", + " \"training_task_inputs\": task,\n", + " \"input_data_config\": input_config,\n", + " \"model_to_upload\": {\"display_name\": model_name},\n", + " }\n", + "\n", + " try:\n", + " pipeline = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " )\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "task_requirements:automl,ten" + }, + "source": [ + "### Construct the task requirements\n", + "\n", + "Next, construct the task requirements. Unlike other parameters which take a Python (JSON-like) dictionary, the `task` field takes a Google protobuf Struct, which is very similar to a Python dictionary. Use the `json_format.ParseDict` method for the conversion.\n", + "\n", + "The minimal fields you need to specify are:\n", + "\n", + "- `multi_label`: Whether True/False this is a multi-label (vs single) classification.\n", + "- `budget_milli_node_hours`: The maximum time to budget (billed) for training the model, where 1000 = 1 hour.\n", + "- `model_type`: The type of deployed model:\n", + " - `CLOUD`: For deploying to Google Cloud.\n", + "- `disable_early_stopping`: Whether True/False to let AutoML use its judgement to stop training early or train for the entire budget.\n", + "\n", + "Finally, you create the pipeline by calling the helper function `create_pipeline`, which returns an instance of a training pipeline object." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_requirements:automl,ten" + }, + "outputs": [], + "source": [ + "PIPE_NAME = \"biomedical_pipe-\" + TIMESTAMP\n", + "MODEL_NAME = \"biomedical_model-\" + TIMESTAMP\n", + "\n", + "task = json_format.ParseDict(\n", + " {\n", + " \"multi_label\": False,\n", + " \"budget_milli_node_hours\": 8000,\n", + " \"model_type\": \"CLOUD\",\n", + " \"disable_early_stopping\": False,\n", + " },\n", + " Value(),\n", + ")\n", + "\n", + "response = create_pipeline(PIPE_NAME, MODEL_NAME, dataset_id, TRAINING_SCHEMA, task)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 120 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "model_information" + }, + "source": [ + "## Model information\n", + "\n", + "Now that your model is trained, you can get some information on your model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:automl" + }, + "source": [ + "## Evaluate the Model resource\n", + "\n", + "Now find out how good the model service believes your model is. As part of training, some portion of the dataset was set aside as the test (holdout) data, which is used by the pipeline service to evaluate the model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "list_model_evaluations:automl,ten" + }, + "source": [ + "### List evaluations for all slices\n", + "\n", + "Use this helper function `list_model_evaluations`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified model identifier for the `Model` resource.\n", + "\n", + "This helper function uses the model client service's `list_model_evaluations` method, which takes the same parameter. The response object from the call is a list, where each element is an evaluation metric.\n", + "\n", + "For each evaluation -- you probably only have one, we then print all the key names for each metric in the evaluation, and for a small set (`confusionMatrix` and `confidenceMetrics`) you will print the result." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "list_model_evaluations:automl,ten" + }, + "outputs": [], + "source": [ + "def list_model_evaluations(name):\n", + " response = clients[\"model\"].list_model_evaluations(parent=name)\n", + " for evaluation in response:\n", + " print(\"model_evaluation\")\n", + " print(\" name:\", evaluation.name)\n", + " print(\" metrics_schema_uri:\", evaluation.metrics_schema_uri)\n", + " metrics = json_format.MessageToDict(evaluation._pb.metrics)\n", + " for metric in metrics.keys():\n", + " print(metric)\n", + " print(\"confusionMatrix\", metrics[\"confusionMatrix\"])\n", + " print(\"confidenceMetrics\", metrics[\"confidenceMetrics\"])\n", + "\n", + " return evaluation.name\n", + "\n", + "\n", + "last_evaluation = list_model_evaluations(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy:batch_prediction" + }, + "source": [ + "## Model deployment for batch prediction\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created for batch prediction. This differs from deploying a `Model` resource for on-demand prediction.\n", + "\n", + "For online prediction, you:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource.\n", + "\n", + "3. Make online prediction requests to the `Endpoint` resource.\n", + "\n", + "For batch-prediction, you:\n", + "\n", + "1. Create a batch prediction job.\n", + "\n", + "2. The job service will provision resources for the batch prediction request.\n", + "\n", + "3. The results of the batch prediction request are returned to the caller.\n", + "\n", + "4. The job service will unprovision the resoures for the batch prediction request." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a batch prediction request\n", + "\n", + "Now do a batch prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_test_items:automl,batch_prediction" + }, + "source": [ + "### Make test items\n", + "\n", + "You will use synthetic data as a test data items. Don't be concerned that we are using synthetic data -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_test_items:automl,text,biomedical" + }, + "outputs": [], + "source": [ + "test_item_1 = 'Molecular basis of hexosaminidase A deficiency and pseudodeficiency in the Berks County Pennsylvania Dutch.\\tFollowing the birth of two infants with Tay-Sachs disease ( TSD ) , a non-Jewish , Pennsylvania Dutch kindred was screened for TSD carriers using the biochemical assay . A high frequency of individuals who appeared to be TSD heterozygotes was detected ( Kelly et al . , 1975 ) . Clinical and biochemical evidence suggested that the increased carrier frequency was due to at least two altered alleles for the hexosaminidase A alpha-subunit . We now report two mutant alleles in this Pennsylvania Dutch kindred , and one polymorphism . One allele , reported originally in a French TSD patient ( Akli et al . , 1991 ) , is a GT-- > AT transition at the donor splice-site of intron 9 . The second , a C-- > T transition at nucleotide 739 ( Arg247Trp ) , has been shown by Triggs-Raine et al . ( 1992 ) to be a clinically benign \" pseudodeficient \" allele associated with reduced enzyme activity against artificial substrate . Finally , a polymorphism [ G-- > A ( 759 ) ] , which leaves valine at codon 253 unchanged , is described'\n", + "test_item_2 = \"Analysis of alkaptonuria (AKU) mutations and polymorphisms reveals that the CCC sequence motif is a mutational hot spot in the homogentisate 1,2 dioxygenase gene (HGO).\tWe recently showed that alkaptonuria ( AKU ) is caused by loss-of-function mutations in the homogentisate 1 , 2 dioxygenase gene ( HGO ) . Herein we describe haplotype and mutational analyses of HGO in seven new AKU pedigrees . These analyses identified two novel single-nucleotide polymorphisms ( INV4 + 31A-- > G and INV11 + 18A-- > G ) and six novel AKU mutations ( INV1-1G-- > A , W60G , Y62C , A122D , P230T , and D291E ) , which further illustrates the remarkable allelic heterogeneity found in AKU . Reexamination of all 29 mutations and polymorphisms thus far described in HGO shows that these nucleotide changes are not randomly distributed ; the CCC sequence motif and its inverted complement , GGG , are preferentially mutated . These analyses also demonstrated that the nucleotide substitutions in HGO do not involve CpG dinucleotides , which illustrates important differences between HGO and other genes for the occurrence of mutation at specific short-sequence motifs . Because the CCC sequence motifs comprise a significant proportion ( 34 . 5 % ) of all mutated bases that have been observed in HGO , we conclude that the CCC triplet is a mutational hot spot in HGO .\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:automl,text" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "Now make a batch input file, which you will store in your local Cloud Storage bucket. The batch input file can only be in JSONL format. For JSONL file, you make one dictionary entry per line for each data item (instance). The dictionary contains the key/value pairs:\n", + "\n", + "- `content`: The Cloud Storage path to the file with the text item.\n", + "- `mime_type`: The content type. In our example, it is an `text` file.\n", + "\n", + "For example:\n", + "\n", + " {'content': '[your-bucket]/file1.txt', 'mime_type': 'text'}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_file:automl,text" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "import tensorflow as tf\n", + "\n", + "gcs_test_item_1 = BUCKET_NAME + \"/test1.txt\"\n", + "with tf.io.gfile.GFile(gcs_test_item_1, \"w\") as f:\n", + " f.write(test_item_1 + \"\\n\")\n", + "gcs_test_item_2 = BUCKET_NAME + \"/test2.txt\"\n", + "with tf.io.gfile.GFile(gcs_test_item_2, \"w\") as f:\n", + " f.write(test_item_2 + \"\\n\")\n", + "\n", + "gcs_input_uri = BUCKET_NAME + \"/test.jsonl\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " data = {\"content\": gcs_test_item_1, \"mime_type\": \"text/plain\"}\n", + " f.write(json.dumps(data) + \"\\n\")\n", + " data = {\"content\": gcs_test_item_2, \"mime_type\": \"text/plain\"}\n", + " f.write(json.dumps(data) + \"\\n\")\n", + "\n", + "print(gcs_input_uri)\n", + "! gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your batch prediction requests:\n", + "\n", + "- Single Instance: The batch prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The batch prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and batch prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The batch prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_request:automl,ten" + }, + "source": [ + "### Make batch prediction request\n", + "\n", + "Now that your batch of two test items is ready, let's do the batch request. Use this helper function `create_batch_prediction_job`, with the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the prediction job.\n", + "- `model_name`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `gcs_source_uri`: The Cloud Storage path to the input file -- which you created above.\n", + "- `gcs_destination_output_uri_prefix`: The Cloud Storage path that the service will write the predictions to.\n", + "- `parameters`: Additional filtering parameters for serving prediction results.\n", + "\n", + "The helper function calls the job client service's `create_batch_prediction_job` metho, with the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for Dataset, Model and Pipeline resources.\n", + "- `batch_prediction_job`: The specification for the batch prediction job.\n", + "\n", + "Let's now dive into the specification for the `batch_prediction_job`:\n", + "\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `dedicated_resources`: The compute resources to provision for the batch prediction job.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `starting_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "- `model_parameters`: Additional filtering parameters for serving prediction results. *Note*, text models do not support additional parameters.\n", + "- `input_config`: The input source and format type for the instances to predict.\n", + " - `instances_format`: The format of the batch prediction request file: `jsonl` only supported.\n", + " - `gcs_source`: A list of one or more Cloud Storage paths to your batch prediction requests.\n", + "- `output_config`: The output destination and format for the predictions.\n", + " - `prediction_format`: The format of the batch prediction response file: `jsonl` only supported.\n", + " - `gcs_destination`: The output destination for the predictions.\n", + "- `dedicated_resources`: The compute resources to provision for the batch prediction job.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `starting_replica_count`: The number of compute instances to initially provision.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to. In this tutorial, only one instance is provisioned.\n", + "\n", + "This call is an asychronous operation. You will print from the response object a few select fields, including:\n", + "\n", + "- `name`: The Vertex fully qualified identifier assigned to the batch prediction job.\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the Model resource.\n", + "- `generate_explanations`: Whether True/False explanations were provided with the predictions (explainability).\n", + "- `state`: The state of the prediction job (pending, running, etc).\n", + "\n", + "Since this call will take a few moments to execute, you will likely get `JobState.JOB_STATE_PENDING` for `state`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_request:automl,ten" + }, + "outputs": [], + "source": [ + "BATCH_MODEL = \"biomedical_batch-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_batch_prediction_job(\n", + " display_name,\n", + " model_name,\n", + " gcs_source_uri,\n", + " gcs_destination_output_uri_prefix,\n", + " parameters=None,\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " batch_prediction_job = {\n", + " \"display_name\": display_name,\n", + " # Format: 'projects/{project}/locations/{location}/models/{model_id}'\n", + " \"model\": model_name,\n", + " \"model_parameters\": json_format.ParseDict(parameters, Value()),\n", + " \"input_config\": {\n", + " \"instances_format\": IN_FORMAT,\n", + " \"gcs_source\": {\"uris\": [gcs_source_uri]},\n", + " },\n", + " \"output_config\": {\n", + " \"predictions_format\": OUT_FORMAT,\n", + " \"gcs_destination\": {\"output_uri_prefix\": gcs_destination_output_uri_prefix},\n", + " },\n", + " \"dedicated_resources\": {\n", + " \"machine_spec\": machine_spec,\n", + " \"starting_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " },\n", + " }\n", + " response = clients[\"job\"].create_batch_prediction_job(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + " )\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try:\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", response.labels)\n", + " return response\n", + "\n", + "\n", + "IN_FORMAT = \"jsonl\"\n", + "OUT_FORMAT = \"jsonl\" # [jsonl]\n", + "\n", + "response = create_batch_prediction_job(\n", + " BATCH_MODEL, model_to_deploy_id, gcs_input_uri, BUCKET_NAME, None\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batch_job_id:response" + }, + "source": [ + "Now get the unique identifier for the batch prediction job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batch_job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the batch job\n", + "batch_job_id = response.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split(\"/\")[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_batch_prediction_job" + }, + "source": [ + "### Get information on a batch prediction job\n", + "\n", + "Use this helper function `get_batch_prediction_job`, with the following paramter:\n", + "\n", + "- `job_name`: The Vertex fully qualified identifier for the batch prediction job.\n", + "\n", + "The helper function calls the job client service's `get_batch_prediction_job` method, with the following paramter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the batch prediction job. In this tutorial, you will pass it the Vertex fully qualified identifier for your batch prediction job -- `batch_job_id`\n", + "\n", + "The helper function will return the Cloud Storage path to where the predictions are stored -- `gcs_destination`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_batch_prediction_job" + }, + "outputs": [], + "source": [ + "def get_batch_prediction_job(job_name, silent=False):\n", + " response = clients[\"job\"].get_batch_prediction_job(name=job_name)\n", + " if silent:\n", + " return response.output_config.gcs_destination.output_uri_prefix, response.state\n", + "\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try: # not all data types support explanations\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" error:\", response.error)\n", + " gcs_destination = response.output_config.gcs_destination\n", + " print(\" gcs_destination\")\n", + " print(\" output_uri_prefix:\", gcs_destination.output_uri_prefix)\n", + " return gcs_destination.output_uri_prefix, response.state\n", + "\n", + "\n", + "predictions, state = get_batch_prediction_job(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_the_predictions:automl,ten" + }, + "source": [ + "### Get the predictions\n", + "\n", + "When the batch prediction is done processing, the job state will be `JOB_STATE_SUCCEEDED`.\n", + "\n", + "Finally you view the predictions stored at the Cloud Storage path you set as output. The predictions will be in a JSONL format, which you indicated at the time you made the batch prediction job, under a subfolder starting with the name `prediction`, and under that folder will be a file called `predictions*.jsonl`.\n", + "\n", + "Now display (cat) the contents. You will see multiple JSON objects, one for each prediction.\n", + "\n", + "The first field `text_snippet` is the text file you did the prediction on, and the second field `annotations` is the prediction, which is further broken down into:\n", + "\n", + "- `text_extraction`: The extracted entity from the text.\n", + "- `display_name`: The predicted label for the extraction entity.\n", + "- `score`: The confidence level between 0 and 1 in the prediction.\n", + "- `startOffset`: The character offset in the text of the start of the extracted entity.\n", + "- `endOffset`: The character offset in the text of the end of the extracted entity." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_the_predictions:automl,text" + }, + "outputs": [], + "source": [ + "def get_latest_predictions(gcs_out_dir):\n", + " \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n", + " folders = !gsutil ls $gcs_out_dir\n", + " latest = \"\"\n", + " for folder in folders:\n", + " subfolder = folder.split(\"/\")[-2]\n", + " if subfolder.startswith(\"prediction-\"):\n", + " if subfolder > latest:\n", + " latest = folder[:-1]\n", + " return latest\n", + "\n", + "\n", + "while True:\n", + " predictions, state = get_batch_prediction_job(batch_job_id, True)\n", + " if state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", state)\n", + " if state == aip.JobState.JOB_STATE_FAILED:\n", + " raise Exception(\"Batch Job Failed\")\n", + " else:\n", + " folder = get_latest_predictions(predictions)\n", + " ! gsutil ls $folder/prediction*.jsonl\n", + "\n", + " ! gsutil cat $folder/prediction*.jsonl\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_automl_text_entity_extraction_batch.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/automl/showcase_automl_text_entity_extraction_online.ipynb b/notebooks/community/gapic/automl/showcase_automl_text_entity_extraction_online.ipynb new file mode 100644 index 000000000..5fbece574 --- /dev/null +++ b/notebooks/community/gapic/automl/showcase_automl_text_entity_extraction_online.ipynb @@ -0,0 +1,1563 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: AutoML text entity extraction model for online prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to create text entity extraction models and do online prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:biomedical,ten" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [NCBI Disease Research Abstracts dataset](https://www.ncbi.nlm.nih.gov/CBBresearch/Dogan/DISEASE/) from [National Center for Biotechnology Information](https://www.ncbi.nlm.nih.gov/). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,online_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML text entity extraction model and deploy for online prediction from a Python script using the Vertex client library. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model`." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Set constants unique to AutoML datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the `Dataset` resource service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the `Dataset` resource service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:ten" + }, + "outputs": [], + "source": [ + "# Text Dataset type\n", + "DATA_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/metadata/text_1.0.0.yaml\"\n", + "# Text Labeling type\n", + "LABEL_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/ioformat/text_extraction_io_format_1.0.0.yaml\"\n", + "# Text Training task\n", + "TRAINING_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_text_extraction_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML text entity extraction model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:automl,online_prediction" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training.\n", + "- Endpoint Service for deployment.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:automl,online_prediction" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "\n", + "def create_dataset(name, schema, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " dataset = aip.Dataset(\n", + " display_name=name, metadata_schema_uri=schema, labels=labels\n", + " )\n", + "\n", + " operation = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"biomedical-\" + TIMESTAMP, DATA_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:text,ten,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for text has a couple of requirements for your text entity extraction data.\n", + "\n", + "- Text examples must be stored in a JSONL file. Unlike text classification and sentiment analysis, a CSV index file is not supported.\n", + "- The examples must be either inline text or reference text files that are in Cloud Storage buckets." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:ten,u_dataset,jsonl" + }, + "source": [ + "#### JSONL\n", + "\n", + "For text entity extraction, the JSONL file has a few requirements:\n", + "\n", + "- Each data item is a separate JSON object, on a separate line.\n", + "- The key/value pair `text_segment_annotations` is a list of character start/end positions in the text per entity with the corresponding label.\n", + " - `display_name`: The label.\n", + " - `start_offset/end_offset`: The character offsets of the start/end of the entity.\n", + "- The key/value pair `text_content` is the text.\n", + "\n", + " {'text_segment_annotations': [{'end_offset': value, 'start_offset': value, 'display_name': label}, ...], 'text_content': text}\n", + "\n", + "*Note*: The dictionary key fields may alternatively be in camelCase. For example, 'display_name' can also be 'displayName'." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,jsonl" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the JSONL index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:biomedical,jsonl,ten" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://ucaip-test-us-central1/dataset/ucaip_ten_dataset.jsonl\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:jsonl" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the NCBI Biomedical dataset that is stored in a public Cloud Storage bucket, using a JSONL index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of objects in a JSONL index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:jsonl" + }, + "outputs": [], + "source": [ + "if \"IMPORT_FILES\" in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_data" + }, + "source": [ + "### Import data\n", + "\n", + "Now, import the data into your Vertex Dataset resource. Use this helper function `import_data` to import the data. The function does the following:\n", + "\n", + "- Uses the `Dataset` client.\n", + "- Calls the client method `import_data`, with the following parameters:\n", + " - `name`: The human readable name you give to the `Dataset` resource (e.g., biomedical).\n", + " - `import_configs`: The import configuration.\n", + "\n", + "- `import_configs`: A Python list containing a dictionary, with the key/value entries:\n", + " - `gcs_sources`: A list of URIs to the paths of the one or more index files.\n", + " - `import_schema_uri`: The schema identifying the labeling type.\n", + "\n", + "The `import_data()` method returns a long running `operation` object. This will take a few minutes to complete. If you are in a live tutorial, this would be a good time to ask questions, or take a personal break." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_data" + }, + "outputs": [], + "source": [ + "def import_data(dataset, gcs_sources, schema):\n", + " config = [{\"gcs_source\": {\"uris\": gcs_sources}, \"import_schema_uri\": schema}]\n", + " print(\"dataset:\", dataset_id)\n", + " start_time = time.time()\n", + " try:\n", + " operation = clients[\"dataset\"].import_data(\n", + " name=dataset_id, import_configs=config\n", + " )\n", + " print(\"Long running operation:\", operation.operation.name)\n", + "\n", + " result = operation.result()\n", + " print(\"result:\", result)\n", + " print(\"time:\", int(time.time() - start_time), \"secs\")\n", + " print(\"error:\", operation.exception())\n", + " print(\"meta :\", operation.metadata)\n", + " print(\n", + " \"after: running:\",\n", + " operation.running(),\n", + " \"done:\",\n", + " operation.done(),\n", + " \"cancelled:\",\n", + " operation.cancelled(),\n", + " )\n", + "\n", + " return operation\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "import_data(dataset_id, [IMPORT_FILE], LABEL_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML text entity extraction model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_pipeline:automl" + }, + "source": [ + "### Create a training pipeline\n", + "\n", + "You may ask, what do we use a pipeline for? You typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `pipeline_name`: A human readable name for the pipeline job.\n", + "- `model_name`: A human readable name for the model.\n", + "- `dataset`: The Vertex fully qualified dataset identifier.\n", + "- `schema`: The dataset labeling (annotation) training schema.\n", + "- `task`: A dictionary describing the requirements for the training job.\n", + "\n", + "The helper function calls the `Pipeline` client service'smethod `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "Let's look now deeper into the *minimal* requirements for constructing a `training_pipeline` specification:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The dataset labeling (annotation) training schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A human readable name for the model.\n", + "- `input_data_config`: The dataset specification.\n", + " - `dataset_id`: The Vertex dataset identifier only (non-fully qualified) -- this is the last part of the fully-qualified identifier.\n", + " - `fraction_split`: If specified, the percentages of the dataset to use for training, test and validation. Otherwise, the percentages are automatically selected by AutoML." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_pipeline:automl" + }, + "outputs": [], + "source": [ + "def create_pipeline(pipeline_name, model_name, dataset, schema, task):\n", + "\n", + " dataset_id = dataset.split(\"/\")[-1]\n", + "\n", + " input_config = {\n", + " \"dataset_id\": dataset_id,\n", + " \"fraction_split\": {\n", + " \"training_fraction\": 0.8,\n", + " \"validation_fraction\": 0.1,\n", + " \"test_fraction\": 0.1,\n", + " },\n", + " }\n", + "\n", + " training_pipeline = {\n", + " \"display_name\": pipeline_name,\n", + " \"training_task_definition\": schema,\n", + " \"training_task_inputs\": task,\n", + " \"input_data_config\": input_config,\n", + " \"model_to_upload\": {\"display_name\": model_name},\n", + " }\n", + "\n", + " try:\n", + " pipeline = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " )\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "task_requirements:automl,ten" + }, + "source": [ + "### Construct the task requirements\n", + "\n", + "Next, construct the task requirements. Unlike other parameters which take a Python (JSON-like) dictionary, the `task` field takes a Google protobuf Struct, which is very similar to a Python dictionary. Use the `json_format.ParseDict` method for the conversion.\n", + "\n", + "The minimal fields you need to specify are:\n", + "\n", + "- `multi_label`: Whether True/False this is a multi-label (vs single) classification.\n", + "- `budget_milli_node_hours`: The maximum time to budget (billed) for training the model, where 1000 = 1 hour.\n", + "- `model_type`: The type of deployed model:\n", + " - `CLOUD`: For deploying to Google Cloud.\n", + "- `disable_early_stopping`: Whether True/False to let AutoML use its judgement to stop training early or train for the entire budget.\n", + "\n", + "Finally, you create the pipeline by calling the helper function `create_pipeline`, which returns an instance of a training pipeline object." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_requirements:automl,ten" + }, + "outputs": [], + "source": [ + "PIPE_NAME = \"biomedical_pipe-\" + TIMESTAMP\n", + "MODEL_NAME = \"biomedical_model-\" + TIMESTAMP\n", + "\n", + "task = json_format.ParseDict(\n", + " {\n", + " \"multi_label\": False,\n", + " \"budget_milli_node_hours\": 8000,\n", + " \"model_type\": \"CLOUD\",\n", + " \"disable_early_stopping\": False,\n", + " },\n", + " Value(),\n", + ")\n", + "\n", + "response = create_pipeline(PIPE_NAME, MODEL_NAME, dataset_id, TRAINING_SCHEMA, task)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 120 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "model_information" + }, + "source": [ + "## Model information\n", + "\n", + "Now that your model is trained, you can get some information on your model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:automl" + }, + "source": [ + "## Evaluate the Model resource\n", + "\n", + "Now find out how good the model service believes your model is. As part of training, some portion of the dataset was set aside as the test (holdout) data, which is used by the pipeline service to evaluate the model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "list_model_evaluations:automl,ten" + }, + "source": [ + "### List evaluations for all slices\n", + "\n", + "Use this helper function `list_model_evaluations`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified model identifier for the `Model` resource.\n", + "\n", + "This helper function uses the model client service's `list_model_evaluations` method, which takes the same parameter. The response object from the call is a list, where each element is an evaluation metric.\n", + "\n", + "For each evaluation -- you probably only have one, we then print all the key names for each metric in the evaluation, and for a small set (`confusionMatrix` and `confidenceMetrics`) you will print the result." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "list_model_evaluations:automl,ten" + }, + "outputs": [], + "source": [ + "def list_model_evaluations(name):\n", + " response = clients[\"model\"].list_model_evaluations(parent=name)\n", + " for evaluation in response:\n", + " print(\"model_evaluation\")\n", + " print(\" name:\", evaluation.name)\n", + " print(\" metrics_schema_uri:\", evaluation.metrics_schema_uri)\n", + " metrics = json_format.MessageToDict(evaluation._pb.metrics)\n", + " for metric in metrics.keys():\n", + " print(metric)\n", + " print(\"confusionMatrix\", metrics[\"confusionMatrix\"])\n", + " print(\"confidenceMetrics\", metrics[\"confidenceMetrics\"])\n", + "\n", + " return evaluation.name\n", + "\n", + "\n", + "last_evaluation = list_model_evaluations(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:automl" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created with AutoML. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"biomedical_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:automatic" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `automatic_resources`: This refers to how many redundant compute instances (replicas). For this example, we set it to one (no replication).\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:automatic" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"biomedical_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"automatic_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " },\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_test_item:automl,online_prediction" + }, + "source": [ + "### Make test item\n", + "\n", + "You will use synthetic data as a test data item. Don't be concerned that we are using synthetic data -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_test_item:automl,text,biomedical" + }, + "outputs": [], + "source": [ + "test_item = 'Molecular basis of hexosaminidase A deficiency and pseudodeficiency in the Berks County Pennsylvania Dutch.\\tFollowing the birth of two infants with Tay-Sachs disease ( TSD ) , a non-Jewish , Pennsylvania Dutch kindred was screened for TSD carriers using the biochemical assay . A high frequency of individuals who appeared to be TSD heterozygotes was detected ( Kelly et al . , 1975 ) . Clinical and biochemical evidence suggested that the increased carrier frequency was due to at least two altered alleles for the hexosaminidase A alpha-subunit . We now report two mutant alleles in this Pennsylvania Dutch kindred , and one polymorphism . One allele , reported originally in a French TSD patient ( Akli et al . , 1991 ) , is a GT-- > AT transition at the donor splice-site of intron 9 . The second , a C-- > T transition at nucleotide 739 ( Arg247Trp ) , has been shown by Triggs-Raine et al . ( 1992 ) to be a clinically benign \" pseudodeficient \" allele associated with reduced enzyme activity against artificial substrate . Finally , a polymorphism [ G-- > A ( 759 ) ] , which leaves valine at codon 253 unchanged , is described'" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "predict_item:automl,ten" + }, + "source": [ + "### Make a prediction\n", + "\n", + "Now you have a test item. Use this helper function `predict_item`, which takes the following parameters:\n", + "\n", + "- `filename`: The Cloud Storage path to the test item.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `parameters_dict`: Additional filtering parameters for serving prediction results.\n", + "\n", + "This function calls the prediction client service's `predict` method with the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `instances`: A list of instances (text files) to predict.\n", + "- `parameters`: Additional filtering parameters for serving prediction results. *Note*, text models do not support additional parameters.\n", + "\n", + "#### Request\n", + "\n", + "The format of each instance is:\n", + "\n", + " { 'content': text_item }\n", + "\n", + "Since the `predict()` method can take multiple items (instances), you send your single test item as a list of one test item. As a final step, you package the instances list into Google's protobuf format -- which is what we pass to the `predict()` method.\n", + "\n", + "#### Response\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding data item in the request. You will see in the output for each prediction -- in our case there is just one:\n", + "\n", + "- `prediction`: A list of IDs assigned to each entity extracted from the text.\n", + "- `confidences`: The confidence level between 0 and 1 for each entity.\n", + "- `display_names`: The label name for each entity.\n", + "- `textSegmentStartOffsets`: The character start location of the entity in the text.\n", + "- `textSegmentEndOffsets`: The character end location of the entity in the text." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "predict_item:automl,ten" + }, + "outputs": [], + "source": [ + "def predict_item(data, endpoint, parameters_dict):\n", + "\n", + " parameters = json_format.ParseDict(parameters_dict, Value())\n", + "\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [{\"content\": data}]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", dict(prediction))\n", + " return response\n", + "\n", + "\n", + "response = predict_item(test_item, endpoint_id, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_automl_text_entity_extraction_online.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/automl/showcase_automl_text_multi-label_classification_batch.ipynb b/notebooks/community/gapic/automl/showcase_automl_text_multi-label_classification_batch.ipynb new file mode 100644 index 000000000..e178eb25f --- /dev/null +++ b/notebooks/community/gapic/automl/showcase_automl_text_multi-label_classification_batch.ipynb @@ -0,0 +1,1779 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: AutoML text multi-label classification model for batch prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to create text multi-label classification models and do batch prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:mcdonalds,tmcn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [McDonald's Service](https://TODO). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,batch_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML text multi-label classification model from a Python script, and then do a batch prediction using the Vertex client library. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Make a batch prediction.\n", + "\n", + "There is one key difference between using batch prediction and using online prediction:\n", + "\n", + "* Prediction Service: Does an on-demand prediction for the entire set of instances (i.e., one or more data items) and returns the results in real-time.\n", + "\n", + "* Batch Prediction Service: Does a queued (batch) prediction for the entire set of instances in the background and stores the results in a Cloud Storage bucket when ready." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Set constants unique to AutoML datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the `Dataset` resource service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the `Dataset` resource service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:tmcn" + }, + "outputs": [], + "source": [ + "# Text Dataset type\n", + "DATA_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/metadata/text_1.0.0.yaml\"\n", + "# Text Labeling type\n", + "LABEL_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/ioformat/text_classification_multi_label_io_format_1.0.0.yaml\"\n", + "# Text Training task\n", + "TRAINING_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_text_classification_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:prediction" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for prediction.\n", + "\n", + "Set the variable `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:automl" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "For AutoML batch prediction, the container image for the serving binary is pre-determined by the Vertex prediction service. More specifically, the service will pick the appropriate container for the model depending on the hardware accelerator you selected." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for prediction.\n", + "\n", + "- Set the variable `DEPLOY_COMPUTE` to configure the compute resources for the VM you will use for prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML text multi-label classification model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:automl,batch_prediction" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training.\n", + "- Job Service for batch prediction and custom training." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:automl,batch_prediction" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"job\"] = create_job_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "\n", + "def create_dataset(name, schema, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " dataset = aip.Dataset(\n", + " display_name=name, metadata_schema_uri=schema, labels=labels\n", + " )\n", + "\n", + " operation = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"mcdonalds-\" + TIMESTAMP, DATA_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:text,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for text has a couple of requirements for your text data.\n", + "\n", + "- Text examples must be stored in a CSV or JSONL file." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:tmcn,u_dataset,csv" + }, + "source": [ + "#### CSV\n", + "\n", + "For text multi-label classification, the CSV file has a few requirements:\n", + "\n", + "- No heading.\n", + "- First column is the text example.\n", + "- Remaining columns are the labels." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:mcdonalds,csv,tmcn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://ucaip-test-us-central1/dataset/ucaip_multi_tcn_dataset.csv\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:csv" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the McDonald's Service dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:csv" + }, + "outputs": [], + "source": [ + "if \"IMPORT_FILES\" in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_data" + }, + "source": [ + "### Import data\n", + "\n", + "Now, import the data into your Vertex Dataset resource. Use this helper function `import_data` to import the data. The function does the following:\n", + "\n", + "- Uses the `Dataset` client.\n", + "- Calls the client method `import_data`, with the following parameters:\n", + " - `name`: The human readable name you give to the `Dataset` resource (e.g., mcdonalds).\n", + " - `import_configs`: The import configuration.\n", + "\n", + "- `import_configs`: A Python list containing a dictionary, with the key/value entries:\n", + " - `gcs_sources`: A list of URIs to the paths of the one or more index files.\n", + " - `import_schema_uri`: The schema identifying the labeling type.\n", + "\n", + "The `import_data()` method returns a long running `operation` object. This will take a few minutes to complete. If you are in a live tutorial, this would be a good time to ask questions, or take a personal break." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_data" + }, + "outputs": [], + "source": [ + "def import_data(dataset, gcs_sources, schema):\n", + " config = [{\"gcs_source\": {\"uris\": gcs_sources}, \"import_schema_uri\": schema}]\n", + " print(\"dataset:\", dataset_id)\n", + " start_time = time.time()\n", + " try:\n", + " operation = clients[\"dataset\"].import_data(\n", + " name=dataset_id, import_configs=config\n", + " )\n", + " print(\"Long running operation:\", operation.operation.name)\n", + "\n", + " result = operation.result()\n", + " print(\"result:\", result)\n", + " print(\"time:\", int(time.time() - start_time), \"secs\")\n", + " print(\"error:\", operation.exception())\n", + " print(\"meta :\", operation.metadata)\n", + " print(\n", + " \"after: running:\",\n", + " operation.running(),\n", + " \"done:\",\n", + " operation.done(),\n", + " \"cancelled:\",\n", + " operation.cancelled(),\n", + " )\n", + "\n", + " return operation\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "import_data(dataset_id, [IMPORT_FILE], LABEL_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML text multi-label classification model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_pipeline:automl" + }, + "source": [ + "### Create a training pipeline\n", + "\n", + "You may ask, what do we use a pipeline for? You typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `pipeline_name`: A human readable name for the pipeline job.\n", + "- `model_name`: A human readable name for the model.\n", + "- `dataset`: The Vertex fully qualified dataset identifier.\n", + "- `schema`: The dataset labeling (annotation) training schema.\n", + "- `task`: A dictionary describing the requirements for the training job.\n", + "\n", + "The helper function calls the `Pipeline` client service'smethod `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "Let's look now deeper into the *minimal* requirements for constructing a `training_pipeline` specification:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The dataset labeling (annotation) training schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A human readable name for the model.\n", + "- `input_data_config`: The dataset specification.\n", + " - `dataset_id`: The Vertex dataset identifier only (non-fully qualified) -- this is the last part of the fully-qualified identifier.\n", + " - `fraction_split`: If specified, the percentages of the dataset to use for training, test and validation. Otherwise, the percentages are automatically selected by AutoML." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_pipeline:automl" + }, + "outputs": [], + "source": [ + "def create_pipeline(pipeline_name, model_name, dataset, schema, task):\n", + "\n", + " dataset_id = dataset.split(\"/\")[-1]\n", + "\n", + " input_config = {\n", + " \"dataset_id\": dataset_id,\n", + " \"fraction_split\": {\n", + " \"training_fraction\": 0.8,\n", + " \"validation_fraction\": 0.1,\n", + " \"test_fraction\": 0.1,\n", + " },\n", + " }\n", + "\n", + " training_pipeline = {\n", + " \"display_name\": pipeline_name,\n", + " \"training_task_definition\": schema,\n", + " \"training_task_inputs\": task,\n", + " \"input_data_config\": input_config,\n", + " \"model_to_upload\": {\"display_name\": model_name},\n", + " }\n", + "\n", + " try:\n", + " pipeline = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " )\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "task_requirements:automl,tmcn" + }, + "source": [ + "### Construct the task requirements\n", + "\n", + "Next, construct the task requirements. Unlike other parameters which take a Python (JSON-like) dictionary, the `task` field takes a Google protobuf Struct, which is very similar to a Python dictionary. Use the `json_format.ParseDict` method for the conversion.\n", + "\n", + "The minimal fields you need to specify are:\n", + "\n", + "- `multi_label`: Whether True/False this is a multi-label (vs single) classification.\n", + "\n", + "Finally, create the pipeline by calling the helper function `create_pipeline`, which returns an instance of a training pipeline object." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_requirements:automl,tmcn" + }, + "outputs": [], + "source": [ + "PIPE_NAME = \"mcdonalds_pipe-\" + TIMESTAMP\n", + "MODEL_NAME = \"mcdonalds_model-\" + TIMESTAMP\n", + "\n", + "task = json_format.ParseDict(\n", + " {\n", + " \"multi_label\": True,\n", + " },\n", + " Value(),\n", + ")\n", + "\n", + "response = create_pipeline(PIPE_NAME, MODEL_NAME, dataset_id, TRAINING_SCHEMA, task)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 240 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "model_information" + }, + "source": [ + "## Model information\n", + "\n", + "Now that your model is trained, you can get some information on your model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:automl" + }, + "source": [ + "## Evaluate the Model resource\n", + "\n", + "Now find out how good the model service believes your model is. As part of training, some portion of the dataset was set aside as the test (holdout) data, which is used by the pipeline service to evaluate the model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "list_model_evaluations:automl,tmcn" + }, + "source": [ + "### List evaluations for all slices\n", + "\n", + "Use this helper function `list_model_evaluations`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified model identifier for the `Model` resource.\n", + "\n", + "This helper function uses the model client service's `list_model_evaluations` method, which takes the same parameter. The response object from the call is a list, where each element is an evaluation metric.\n", + "\n", + "For each evaluation (you probably only have one) we then print all the key names for each metric in the evaluation, and for a small set (`logLoss` and `auPrc`) you will print the result." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "list_model_evaluations:automl,tmcn" + }, + "outputs": [], + "source": [ + "def list_model_evaluations(name):\n", + " response = clients[\"model\"].list_model_evaluations(parent=name)\n", + " for evaluation in response:\n", + " print(\"model_evaluation\")\n", + " print(\" name:\", evaluation.name)\n", + " print(\" metrics_schema_uri:\", evaluation.metrics_schema_uri)\n", + " metrics = json_format.MessageToDict(evaluation._pb.metrics)\n", + " for metric in metrics.keys():\n", + " print(metric)\n", + " print(\"logloss\", metrics[\"logLoss\"])\n", + " print(\"auPrc\", metrics[\"auPrc\"])\n", + "\n", + " return evaluation.name\n", + "\n", + "\n", + "last_evaluation = list_model_evaluations(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy:batch_prediction" + }, + "source": [ + "## Model deployment for batch prediction\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created for batch prediction. This differs from deploying a `Model` resource for on-demand prediction.\n", + "\n", + "For online prediction, you:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource.\n", + "\n", + "3. Make online prediction requests to the `Endpoint` resource.\n", + "\n", + "For batch-prediction, you:\n", + "\n", + "1. Create a batch prediction job.\n", + "\n", + "2. The job service will provision resources for the batch prediction request.\n", + "\n", + "3. The results of the batch prediction request are returned to the caller.\n", + "\n", + "4. The job service will unprovision the resoures for the batch prediction request." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a batch prediction request\n", + "\n", + "Now do a batch prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_items:batch_prediction" + }, + "source": [ + "### Get test item(s)\n", + "\n", + "Now do a batch prediction to your Vertex model. You will use arbitrary examples out of the dataset as a test items. Don't be concerned that the examples were likely used in training the model -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_items:automl,tmcn,csv" + }, + "outputs": [], + "source": [ + "test_items = ! gsutil cat $IMPORT_FILE | head -n2\n", + "\n", + "cols_1 = str(test_items[0]).split(\",\")\n", + "cols_2 = str(test_items[1]).split(\",\")\n", + "test_item_1 = cols_1[0]\n", + "test_label_1 = cols_1[1:]\n", + "test_item_2 = cols_2[0]\n", + "test_label_2 = cols_2[1:]\n", + "\n", + "print(test_item_1, test_label_1)\n", + "print(test_item_2, test_label_2)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:automl,text" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "Now make a batch input file, which you will store in your local Cloud Storage bucket. The batch input file can only be in JSONL format. For JSONL file, you make one dictionary entry per line for each data item (instance). The dictionary contains the key/value pairs:\n", + "\n", + "- `content`: The Cloud Storage path to the file with the text item.\n", + "- `mime_type`: The content type. In our example, it is an `text` file.\n", + "\n", + "For example:\n", + "\n", + " {'content': '[your-bucket]/file1.txt', 'mime_type': 'text'}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_file:automl,text" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "import tensorflow as tf\n", + "\n", + "gcs_test_item_1 = BUCKET_NAME + \"/test1.txt\"\n", + "with tf.io.gfile.GFile(gcs_test_item_1, \"w\") as f:\n", + " f.write(test_item_1 + \"\\n\")\n", + "gcs_test_item_2 = BUCKET_NAME + \"/test2.txt\"\n", + "with tf.io.gfile.GFile(gcs_test_item_2, \"w\") as f:\n", + " f.write(test_item_2 + \"\\n\")\n", + "\n", + "gcs_input_uri = BUCKET_NAME + \"/test.jsonl\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " data = {\"content\": gcs_test_item_1, \"mime_type\": \"text/plain\"}\n", + " f.write(json.dumps(data) + \"\\n\")\n", + " data = {\"content\": gcs_test_item_2, \"mime_type\": \"text/plain\"}\n", + " f.write(json.dumps(data) + \"\\n\")\n", + "\n", + "print(gcs_input_uri)\n", + "! gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your batch prediction requests:\n", + "\n", + "- Single Instance: The batch prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The batch prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and batch prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The batch prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_request:automl,tmcn" + }, + "source": [ + "### Make batch prediction request\n", + "\n", + "Now that your batch of two test items is ready, let's do the batch request. Use this helper function `create_batch_prediction_job`, with the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the prediction job.\n", + "- `model_name`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `gcs_source_uri`: The Cloud Storage path to the input file -- which you created above.\n", + "- `gcs_destination_output_uri_prefix`: The Cloud Storage path that the service will write the predictions to.\n", + "- `parameters`: Additional filtering parameters for serving prediction results.\n", + "\n", + "The helper function calls the job client service's `create_batch_prediction_job` metho, with the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for Dataset, Model and Pipeline resources.\n", + "- `batch_prediction_job`: The specification for the batch prediction job.\n", + "\n", + "Let's now dive into the specification for the `batch_prediction_job`:\n", + "\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `dedicated_resources`: The compute resources to provision for the batch prediction job.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `starting_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "- `model_parameters`: Additional filtering parameters for serving prediction results. *Note*, text models do not support additional parameters.\n", + "- `input_config`: The input source and format type for the instances to predict.\n", + " - `instances_format`: The format of the batch prediction request file: `jsonl` only supported.\n", + " - `gcs_source`: A list of one or more Cloud Storage paths to your batch prediction requests.\n", + "- `output_config`: The output destination and format for the predictions.\n", + " - `prediction_format`: The format of the batch prediction response file: `jsonl` only supported.\n", + " - `gcs_destination`: The output destination for the predictions.\n", + "- `dedicated_resources`: The compute resources to provision for the batch prediction job.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `starting_replica_count`: The number of compute instances to initially provision.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to. In this tutorial, only one instance is provisioned.\n", + "\n", + "This call is an asychronous operation. You will print from the response object a few select fields, including:\n", + "\n", + "- `name`: The Vertex fully qualified identifier assigned to the batch prediction job.\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the Model resource.\n", + "- `generate_explanations`: Whether True/False explanations were provided with the predictions (explainability).\n", + "- `state`: The state of the prediction job (pending, running, etc).\n", + "\n", + "Since this call will take a few moments to execute, you will likely get `JobState.JOB_STATE_PENDING` for `state`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_request:automl,tmcn" + }, + "outputs": [], + "source": [ + "BATCH_MODEL = \"mcdonalds_batch-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_batch_prediction_job(\n", + " display_name,\n", + " model_name,\n", + " gcs_source_uri,\n", + " gcs_destination_output_uri_prefix,\n", + " parameters=None,\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " batch_prediction_job = {\n", + " \"display_name\": display_name,\n", + " # Format: 'projects/{project}/locations/{location}/models/{model_id}'\n", + " \"model\": model_name,\n", + " \"model_parameters\": json_format.ParseDict(parameters, Value()),\n", + " \"input_config\": {\n", + " \"instances_format\": IN_FORMAT,\n", + " \"gcs_source\": {\"uris\": [gcs_source_uri]},\n", + " },\n", + " \"output_config\": {\n", + " \"predictions_format\": OUT_FORMAT,\n", + " \"gcs_destination\": {\"output_uri_prefix\": gcs_destination_output_uri_prefix},\n", + " },\n", + " \"dedicated_resources\": {\n", + " \"machine_spec\": machine_spec,\n", + " \"starting_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " },\n", + " }\n", + " response = clients[\"job\"].create_batch_prediction_job(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + " )\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try:\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", response.labels)\n", + " return response\n", + "\n", + "\n", + "IN_FORMAT = \"jsonl\"\n", + "OUT_FORMAT = \"jsonl\" # [jsonl]\n", + "\n", + "response = create_batch_prediction_job(\n", + " BATCH_MODEL, model_to_deploy_id, gcs_input_uri, BUCKET_NAME, None\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batch_job_id:response" + }, + "source": [ + "Now get the unique identifier for the batch prediction job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batch_job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the batch job\n", + "batch_job_id = response.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split(\"/\")[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_batch_prediction_job" + }, + "source": [ + "### Get information on a batch prediction job\n", + "\n", + "Use this helper function `get_batch_prediction_job`, with the following paramter:\n", + "\n", + "- `job_name`: The Vertex fully qualified identifier for the batch prediction job.\n", + "\n", + "The helper function calls the job client service's `get_batch_prediction_job` method, with the following paramter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the batch prediction job. In this tutorial, you will pass it the Vertex fully qualified identifier for your batch prediction job -- `batch_job_id`\n", + "\n", + "The helper function will return the Cloud Storage path to where the predictions are stored -- `gcs_destination`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_batch_prediction_job" + }, + "outputs": [], + "source": [ + "def get_batch_prediction_job(job_name, silent=False):\n", + " response = clients[\"job\"].get_batch_prediction_job(name=job_name)\n", + " if silent:\n", + " return response.output_config.gcs_destination.output_uri_prefix, response.state\n", + "\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try: # not all data types support explanations\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" error:\", response.error)\n", + " gcs_destination = response.output_config.gcs_destination\n", + " print(\" gcs_destination\")\n", + " print(\" output_uri_prefix:\", gcs_destination.output_uri_prefix)\n", + " return gcs_destination.output_uri_prefix, response.state\n", + "\n", + "\n", + "predictions, state = get_batch_prediction_job(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_the_predictions:automl,tmcn" + }, + "source": [ + "### Get the predictions\n", + "\n", + "When the batch prediction is done processing, the job state will be `JOB_STATE_SUCCEEDED`.\n", + "\n", + "Finally you view the predictions stored at the Cloud Storage path you set as output. The predictions will be in a JSONL format, which you indicated at the time you made the batch prediction job, under a subfolder starting with the name `prediction`, and under that folder will be a file called `predictions*.jsonl`.\n", + "\n", + "Now display (cat) the contents. You will see multiple JSON objects, one for each prediction.\n", + "\n", + "The first field `text_snippet` is the text file you did the prediction on, and the second field `annotations` is the prediction, which is further broken down into:\n", + "\n", + "- `score`: The percent of confidence between 0 and 1.\n", + "- `display_name`: The corresponding class name." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_the_predictions:automl,text" + }, + "outputs": [], + "source": [ + "def get_latest_predictions(gcs_out_dir):\n", + " \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n", + " folders = !gsutil ls $gcs_out_dir\n", + " latest = \"\"\n", + " for folder in folders:\n", + " subfolder = folder.split(\"/\")[-2]\n", + " if subfolder.startswith(\"prediction-\"):\n", + " if subfolder > latest:\n", + " latest = folder[:-1]\n", + " return latest\n", + "\n", + "\n", + "while True:\n", + " predictions, state = get_batch_prediction_job(batch_job_id, True)\n", + " if state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", state)\n", + " if state == aip.JobState.JOB_STATE_FAILED:\n", + " raise Exception(\"Batch Job Failed\")\n", + " else:\n", + " folder = get_latest_predictions(predictions)\n", + " ! gsutil ls $folder/prediction*.jsonl\n", + "\n", + " ! gsutil cat $folder/prediction*.jsonl\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_automl_text_multi-label_classification_batch.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/automl/showcase_automl_text_multi-label_classification_online.ipynb b/notebooks/community/gapic/automl/showcase_automl_text_multi-label_classification_online.ipynb new file mode 100644 index 000000000..2ffa1b88c --- /dev/null +++ b/notebooks/community/gapic/automl/showcase_automl_text_multi-label_classification_online.ipynb @@ -0,0 +1,1552 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: AutoML text multi-label classification model for online prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to create text multi-label classification models and do online prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:mcdonalds,tmcn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [McDonald's Service](https://TODO). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,online_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML text multi-label classification model and deploy for online prediction from a Python script using the Vertex client library. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model`." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Set constants unique to AutoML datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the `Dataset` resource service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the `Dataset` resource service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:tmcn" + }, + "outputs": [], + "source": [ + "# Text Dataset type\n", + "DATA_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/metadata/text_1.0.0.yaml\"\n", + "# Text Labeling type\n", + "LABEL_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/ioformat/text_classification_multi_label_io_format_1.0.0.yaml\"\n", + "# Text Training task\n", + "TRAINING_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_text_classification_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML text multi-label classification model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:automl,online_prediction" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training.\n", + "- Endpoint Service for deployment.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:automl,online_prediction" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "\n", + "def create_dataset(name, schema, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " dataset = aip.Dataset(\n", + " display_name=name, metadata_schema_uri=schema, labels=labels\n", + " )\n", + "\n", + " operation = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"mcdonalds-\" + TIMESTAMP, DATA_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:text,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for text has a couple of requirements for your text data.\n", + "\n", + "- Text examples must be stored in a CSV or JSONL file." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:tmcn,u_dataset,csv" + }, + "source": [ + "#### CSV\n", + "\n", + "For text multi-label classification, the CSV file has a few requirements:\n", + "\n", + "- No heading.\n", + "- First column is the text example.\n", + "- Remaining columns are the labels." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:mcdonalds,csv,tmcn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://ucaip-test-us-central1/dataset/ucaip_multi_tcn_dataset.csv\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:csv" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the McDonald's Service dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:csv" + }, + "outputs": [], + "source": [ + "if \"IMPORT_FILES\" in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_data" + }, + "source": [ + "### Import data\n", + "\n", + "Now, import the data into your Vertex Dataset resource. Use this helper function `import_data` to import the data. The function does the following:\n", + "\n", + "- Uses the `Dataset` client.\n", + "- Calls the client method `import_data`, with the following parameters:\n", + " - `name`: The human readable name you give to the `Dataset` resource (e.g., mcdonalds).\n", + " - `import_configs`: The import configuration.\n", + "\n", + "- `import_configs`: A Python list containing a dictionary, with the key/value entries:\n", + " - `gcs_sources`: A list of URIs to the paths of the one or more index files.\n", + " - `import_schema_uri`: The schema identifying the labeling type.\n", + "\n", + "The `import_data()` method returns a long running `operation` object. This will take a few minutes to complete. If you are in a live tutorial, this would be a good time to ask questions, or take a personal break." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_data" + }, + "outputs": [], + "source": [ + "def import_data(dataset, gcs_sources, schema):\n", + " config = [{\"gcs_source\": {\"uris\": gcs_sources}, \"import_schema_uri\": schema}]\n", + " print(\"dataset:\", dataset_id)\n", + " start_time = time.time()\n", + " try:\n", + " operation = clients[\"dataset\"].import_data(\n", + " name=dataset_id, import_configs=config\n", + " )\n", + " print(\"Long running operation:\", operation.operation.name)\n", + "\n", + " result = operation.result()\n", + " print(\"result:\", result)\n", + " print(\"time:\", int(time.time() - start_time), \"secs\")\n", + " print(\"error:\", operation.exception())\n", + " print(\"meta :\", operation.metadata)\n", + " print(\n", + " \"after: running:\",\n", + " operation.running(),\n", + " \"done:\",\n", + " operation.done(),\n", + " \"cancelled:\",\n", + " operation.cancelled(),\n", + " )\n", + "\n", + " return operation\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "import_data(dataset_id, [IMPORT_FILE], LABEL_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML text multi-label classification model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_pipeline:automl" + }, + "source": [ + "### Create a training pipeline\n", + "\n", + "You may ask, what do we use a pipeline for? You typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `pipeline_name`: A human readable name for the pipeline job.\n", + "- `model_name`: A human readable name for the model.\n", + "- `dataset`: The Vertex fully qualified dataset identifier.\n", + "- `schema`: The dataset labeling (annotation) training schema.\n", + "- `task`: A dictionary describing the requirements for the training job.\n", + "\n", + "The helper function calls the `Pipeline` client service'smethod `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "Let's look now deeper into the *minimal* requirements for constructing a `training_pipeline` specification:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The dataset labeling (annotation) training schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A human readable name for the model.\n", + "- `input_data_config`: The dataset specification.\n", + " - `dataset_id`: The Vertex dataset identifier only (non-fully qualified) -- this is the last part of the fully-qualified identifier.\n", + " - `fraction_split`: If specified, the percentages of the dataset to use for training, test and validation. Otherwise, the percentages are automatically selected by AutoML." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_pipeline:automl" + }, + "outputs": [], + "source": [ + "def create_pipeline(pipeline_name, model_name, dataset, schema, task):\n", + "\n", + " dataset_id = dataset.split(\"/\")[-1]\n", + "\n", + " input_config = {\n", + " \"dataset_id\": dataset_id,\n", + " \"fraction_split\": {\n", + " \"training_fraction\": 0.8,\n", + " \"validation_fraction\": 0.1,\n", + " \"test_fraction\": 0.1,\n", + " },\n", + " }\n", + "\n", + " training_pipeline = {\n", + " \"display_name\": pipeline_name,\n", + " \"training_task_definition\": schema,\n", + " \"training_task_inputs\": task,\n", + " \"input_data_config\": input_config,\n", + " \"model_to_upload\": {\"display_name\": model_name},\n", + " }\n", + "\n", + " try:\n", + " pipeline = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " )\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "task_requirements:automl,tmcn" + }, + "source": [ + "### Construct the task requirements\n", + "\n", + "Next, construct the task requirements. Unlike other parameters which take a Python (JSON-like) dictionary, the `task` field takes a Google protobuf Struct, which is very similar to a Python dictionary. Use the `json_format.ParseDict` method for the conversion.\n", + "\n", + "The minimal fields you need to specify are:\n", + "\n", + "- `multi_label`: Whether True/False this is a multi-label (vs single) classification.\n", + "\n", + "Finally, create the pipeline by calling the helper function `create_pipeline`, which returns an instance of a training pipeline object." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_requirements:automl,tmcn" + }, + "outputs": [], + "source": [ + "PIPE_NAME = \"mcdonalds_pipe-\" + TIMESTAMP\n", + "MODEL_NAME = \"mcdonalds_model-\" + TIMESTAMP\n", + "\n", + "task = json_format.ParseDict(\n", + " {\n", + " \"multi_label\": True,\n", + " },\n", + " Value(),\n", + ")\n", + "\n", + "response = create_pipeline(PIPE_NAME, MODEL_NAME, dataset_id, TRAINING_SCHEMA, task)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 240 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "model_information" + }, + "source": [ + "## Model information\n", + "\n", + "Now that your model is trained, you can get some information on your model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:automl" + }, + "source": [ + "## Evaluate the Model resource\n", + "\n", + "Now find out how good the model service believes your model is. As part of training, some portion of the dataset was set aside as the test (holdout) data, which is used by the pipeline service to evaluate the model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "list_model_evaluations:automl,tmcn" + }, + "source": [ + "### List evaluations for all slices\n", + "\n", + "Use this helper function `list_model_evaluations`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified model identifier for the `Model` resource.\n", + "\n", + "This helper function uses the model client service's `list_model_evaluations` method, which takes the same parameter. The response object from the call is a list, where each element is an evaluation metric.\n", + "\n", + "For each evaluation (you probably only have one) we then print all the key names for each metric in the evaluation, and for a small set (`logLoss` and `auPrc`) you will print the result." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "list_model_evaluations:automl,tmcn" + }, + "outputs": [], + "source": [ + "def list_model_evaluations(name):\n", + " response = clients[\"model\"].list_model_evaluations(parent=name)\n", + " for evaluation in response:\n", + " print(\"model_evaluation\")\n", + " print(\" name:\", evaluation.name)\n", + " print(\" metrics_schema_uri:\", evaluation.metrics_schema_uri)\n", + " metrics = json_format.MessageToDict(evaluation._pb.metrics)\n", + " for metric in metrics.keys():\n", + " print(metric)\n", + " print(\"logloss\", metrics[\"logLoss\"])\n", + " print(\"auPrc\", metrics[\"auPrc\"])\n", + "\n", + " return evaluation.name\n", + "\n", + "\n", + "last_evaluation = list_model_evaluations(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:automl" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created with AutoML. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"mcdonalds_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:automatic" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `automatic_resources`: This refers to how many redundant compute instances (replicas). For this example, we set it to one (no replication).\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:automatic" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"mcdonalds_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"automatic_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " },\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_item" + }, + "source": [ + "### Get test item\n", + "\n", + "You will use an arbitrary example out of the dataset as a test item. Don't be concerned that the example was likely used in training the model -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:automl,tmcn,csv" + }, + "outputs": [], + "source": [ + "test_item = ! gsutil cat $IMPORT_FILE | head -n1\n", + "\n", + "cols = str(test_item[0]).split(\",\")\n", + "test_item = cols[0]\n", + "test_label = cols[1:]\n", + "\n", + "print(test_item, test_label)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "predict_item:automl,tmcn" + }, + "source": [ + "### Make a prediction\n", + "\n", + "Now you have a test item. Use this helper function `predict_item`, which takes the following parameters:\n", + "\n", + "- `filename`: The Cloud Storage path to the test item.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `parameters_dict`: Additional filtering parameters for serving prediction results.\n", + "\n", + "This function calls the prediction client service's `predict` method with the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `instances`: A list of instances (text files) to predict.\n", + "- `parameters`: Additional filtering parameters for serving prediction results. *Note*, text models do not support additional parameters.\n", + "\n", + "#### Request\n", + "\n", + "The format of each instance is:\n", + "\n", + " { 'content': text_item }\n", + "\n", + "Since the `predict()` method can take multiple items (instances), you send your single test item as a list of one test item. As a final step, you package the instances list into Google's protobuf format -- which is what you pass to the `predict()` method.\n", + "\n", + "#### Response\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding text in the request. You will see in the output for each prediction -- in this case there is just one:\n", + "\n", + "- `confidences`: Confidence level in the prediction.\n", + "- `displayNames`: The predicted label." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "predict_item:automl,tmcn" + }, + "outputs": [], + "source": [ + "def predict_item(data, endpoint, parameters_dict):\n", + "\n", + " parameters = json_format.ParseDict(parameters_dict, Value())\n", + "\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [{\"content\": data}]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", dict(prediction))\n", + " return response\n", + "\n", + "\n", + "response = predict_item(test_item, endpoint_id, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_automl_text_multi-label_classification_online.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/automl/showcase_automl_text_sentiment_analysis_batch.ipynb b/notebooks/community/gapic/automl/showcase_automl_text_sentiment_analysis_batch.ipynb new file mode 100644 index 000000000..b8b3bd50f --- /dev/null +++ b/notebooks/community/gapic/automl/showcase_automl_text_sentiment_analysis_batch.ipynb @@ -0,0 +1,1777 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: AutoML text sentiment analysis model for batch prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to create text sentiment analysis models and do batch prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:claritin,tst" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Crowdflower Claritin-Twitter dataset](https://data.world/crowdflower/claritin-twitter) from [data.world Datasets](https://data.world). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,batch_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML text sentiment analysis model from a Python script, and then do a batch prediction using the Vertex client library. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Make a batch prediction.\n", + "\n", + "There is one key difference between using batch prediction and using online prediction:\n", + "\n", + "* Prediction Service: Does an on-demand prediction for the entire set of instances (i.e., one or more data items) and returns the results in real-time.\n", + "\n", + "* Batch Prediction Service: Does a queued (batch) prediction for the entire set of instances in the background and stores the results in a Cloud Storage bucket when ready." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Set constants unique to AutoML datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the `Dataset` resource service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the `Dataset` resource service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:tst" + }, + "outputs": [], + "source": [ + "# Text Dataset type\n", + "DATA_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/metadata/text_1.0.0.yaml\"\n", + "# Text Labeling type\n", + "LABEL_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/ioformat/text_sentiment_io_format_1.0.0.yaml\"\n", + "# Text Training task\n", + "TRAINING_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_text_sentiment_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:prediction" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for prediction.\n", + "\n", + "Set the variable `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:automl" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "For AutoML batch prediction, the container image for the serving binary is pre-determined by the Vertex prediction service. More specifically, the service will pick the appropriate container for the model depending on the hardware accelerator you selected." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for prediction.\n", + "\n", + "- Set the variable `DEPLOY_COMPUTE` to configure the compute resources for the VM you will use for prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML text sentiment analysis model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:automl,batch_prediction" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training.\n", + "- Job Service for batch prediction and custom training." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:automl,batch_prediction" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"job\"] = create_job_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "\n", + "def create_dataset(name, schema, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " dataset = aip.Dataset(\n", + " display_name=name, metadata_schema_uri=schema, labels=labels\n", + " )\n", + "\n", + " operation = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"claritin-\" + TIMESTAMP, DATA_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:text,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for text has a couple of requirements for your text data.\n", + "\n", + "- Text examples must be stored in a CSV or JSONL file." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:tst,u_dataset,csv" + }, + "source": [ + "#### CSV\n", + "\n", + "For text sentiment analysis, the CSV file has a few requirements:\n", + "\n", + "- No heading.\n", + "- First column is the text example or Cloud Storage path to text file.\n", + "- Second column the label (i.e., sentiment).\n", + "- Third column is the maximum sentiment value. For example, if the range is 0 to 3, then the maximum value is 3." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:claritin,csv,tst" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://cloud-samples-data/language/claritin.csv\"\n", + "SENTIMENT_MAX = 4" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:csv" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Crowdflower Claritin-Twitter dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:csv" + }, + "outputs": [], + "source": [ + "if \"IMPORT_FILES\" in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_data" + }, + "source": [ + "### Import data\n", + "\n", + "Now, import the data into your Vertex Dataset resource. Use this helper function `import_data` to import the data. The function does the following:\n", + "\n", + "- Uses the `Dataset` client.\n", + "- Calls the client method `import_data`, with the following parameters:\n", + " - `name`: The human readable name you give to the `Dataset` resource (e.g., claritin).\n", + " - `import_configs`: The import configuration.\n", + "\n", + "- `import_configs`: A Python list containing a dictionary, with the key/value entries:\n", + " - `gcs_sources`: A list of URIs to the paths of the one or more index files.\n", + " - `import_schema_uri`: The schema identifying the labeling type.\n", + "\n", + "The `import_data()` method returns a long running `operation` object. This will take a few minutes to complete. If you are in a live tutorial, this would be a good time to ask questions, or take a personal break." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_data" + }, + "outputs": [], + "source": [ + "def import_data(dataset, gcs_sources, schema):\n", + " config = [{\"gcs_source\": {\"uris\": gcs_sources}, \"import_schema_uri\": schema}]\n", + " print(\"dataset:\", dataset_id)\n", + " start_time = time.time()\n", + " try:\n", + " operation = clients[\"dataset\"].import_data(\n", + " name=dataset_id, import_configs=config\n", + " )\n", + " print(\"Long running operation:\", operation.operation.name)\n", + "\n", + " result = operation.result()\n", + " print(\"result:\", result)\n", + " print(\"time:\", int(time.time() - start_time), \"secs\")\n", + " print(\"error:\", operation.exception())\n", + " print(\"meta :\", operation.metadata)\n", + " print(\n", + " \"after: running:\",\n", + " operation.running(),\n", + " \"done:\",\n", + " operation.done(),\n", + " \"cancelled:\",\n", + " operation.cancelled(),\n", + " )\n", + "\n", + " return operation\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "import_data(dataset_id, [IMPORT_FILE], LABEL_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML text sentiment analysis model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_pipeline:automl" + }, + "source": [ + "### Create a training pipeline\n", + "\n", + "You may ask, what do we use a pipeline for? You typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `pipeline_name`: A human readable name for the pipeline job.\n", + "- `model_name`: A human readable name for the model.\n", + "- `dataset`: The Vertex fully qualified dataset identifier.\n", + "- `schema`: The dataset labeling (annotation) training schema.\n", + "- `task`: A dictionary describing the requirements for the training job.\n", + "\n", + "The helper function calls the `Pipeline` client service'smethod `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "Let's look now deeper into the *minimal* requirements for constructing a `training_pipeline` specification:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The dataset labeling (annotation) training schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A human readable name for the model.\n", + "- `input_data_config`: The dataset specification.\n", + " - `dataset_id`: The Vertex dataset identifier only (non-fully qualified) -- this is the last part of the fully-qualified identifier.\n", + " - `fraction_split`: If specified, the percentages of the dataset to use for training, test and validation. Otherwise, the percentages are automatically selected by AutoML." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_pipeline:automl" + }, + "outputs": [], + "source": [ + "def create_pipeline(pipeline_name, model_name, dataset, schema, task):\n", + "\n", + " dataset_id = dataset.split(\"/\")[-1]\n", + "\n", + " input_config = {\n", + " \"dataset_id\": dataset_id,\n", + " \"fraction_split\": {\n", + " \"training_fraction\": 0.8,\n", + " \"validation_fraction\": 0.1,\n", + " \"test_fraction\": 0.1,\n", + " },\n", + " }\n", + "\n", + " training_pipeline = {\n", + " \"display_name\": pipeline_name,\n", + " \"training_task_definition\": schema,\n", + " \"training_task_inputs\": task,\n", + " \"input_data_config\": input_config,\n", + " \"model_to_upload\": {\"display_name\": model_name},\n", + " }\n", + "\n", + " try:\n", + " pipeline = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " )\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "task_requirements:automl,tst" + }, + "source": [ + "### Construct the task requirements\n", + "\n", + "Next, construct the task requirements. Unlike other parameters which take a Python (JSON-like) dictionary, the `task` field takes a Google protobuf Struct, which is very similar to a Python dictionary. Use the `json_format.ParseDict` method for the conversion.\n", + "\n", + "The minimal fields we need to specify are:\n", + "\n", + "- `sentiment_max`: The maximum value for the sentiment (e.g., 4).\n", + "\n", + "Finally, create the pipeline by calling the helper function `create_pipeline`, which returns an instance of a training pipeline object." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_requirements:automl,tst" + }, + "outputs": [], + "source": [ + "PIPE_NAME = \"claritin_pipe-\" + TIMESTAMP\n", + "MODEL_NAME = \"claritin_model-\" + TIMESTAMP\n", + "\n", + "task = json_format.ParseDict(\n", + " {\n", + " \"sentiment_max\": SENTIMENT_MAX,\n", + " },\n", + " Value(),\n", + ")\n", + "\n", + "response = create_pipeline(PIPE_NAME, MODEL_NAME, dataset_id, TRAINING_SCHEMA, task)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 180 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "model_information" + }, + "source": [ + "## Model information\n", + "\n", + "Now that your model is trained, you can get some information on your model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:automl" + }, + "source": [ + "## Evaluate the Model resource\n", + "\n", + "Now find out how good the model service believes your model is. As part of training, some portion of the dataset was set aside as the test (holdout) data, which is used by the pipeline service to evaluate the model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "list_model_evaluations:automl,tst" + }, + "source": [ + "### List evaluations for all slices\n", + "\n", + "Use this helper function `list_model_evaluations`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified model identifier for the `Model` resource.\n", + "\n", + "This helper function uses the model client service's `list_model_evaluations` method, which takes the same parameter. The response object from the call is a list, where each element is an evaluation metric.\n", + "\n", + "For each evaluation -- you probably only have one, we then print all the key names for each metric in the evaluation, and for a small set (`meanAbsoluteError` and `precision`) you will print the result." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "list_model_evaluations:automl,tst" + }, + "outputs": [], + "source": [ + "def list_model_evaluations(name):\n", + " response = clients[\"model\"].list_model_evaluations(parent=name)\n", + " for evaluation in response:\n", + " print(\"model_evaluation\")\n", + " print(\" name:\", evaluation.name)\n", + " print(\" metrics_schema_uri:\", evaluation.metrics_schema_uri)\n", + " metrics = json_format.MessageToDict(evaluation._pb.metrics)\n", + " for metric in metrics.keys():\n", + " print(metric)\n", + " print(\"meanAbsoluteError\", metrics[\"meanAbsoluteError\"])\n", + " print(\"precision\", metrics[\"precision\"])\n", + "\n", + " return evaluation.name\n", + "\n", + "\n", + "last_evaluation = list_model_evaluations(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy:batch_prediction" + }, + "source": [ + "## Model deployment for batch prediction\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created for batch prediction. This differs from deploying a `Model` resource for on-demand prediction.\n", + "\n", + "For online prediction, you:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource.\n", + "\n", + "3. Make online prediction requests to the `Endpoint` resource.\n", + "\n", + "For batch-prediction, you:\n", + "\n", + "1. Create a batch prediction job.\n", + "\n", + "2. The job service will provision resources for the batch prediction request.\n", + "\n", + "3. The results of the batch prediction request are returned to the caller.\n", + "\n", + "4. The job service will unprovision the resoures for the batch prediction request." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a batch prediction request\n", + "\n", + "Now do a batch prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_items:batch_prediction" + }, + "source": [ + "### Get test item(s)\n", + "\n", + "Now do a batch prediction to your Vertex model. You will use arbitrary examples out of the dataset as a test items. Don't be concerned that the examples were likely used in training the model -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_items:automl,tst,csv" + }, + "outputs": [], + "source": [ + "test_items = ! gsutil cat $IMPORT_FILE | head -n2\n", + "\n", + "if len(test_items[0]) == 4:\n", + " _, test_item_1, test_label_1, _ = str(test_items[0]).split(\",\")\n", + " _, test_item_2, test_label_2, _ = str(test_items[1]).split(\",\")\n", + "else:\n", + " test_item_1, test_label_1, _ = str(test_items[0]).split(\",\")\n", + " test_item_2, test_label_2, _ = str(test_items[1]).split(\",\")\n", + "\n", + "\n", + "print(test_item_1, test_label_1)\n", + "print(test_item_2, test_label_2)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:automl,text" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "Now make a batch input file, which you will store in your local Cloud Storage bucket. The batch input file can only be in JSONL format. For JSONL file, you make one dictionary entry per line for each data item (instance). The dictionary contains the key/value pairs:\n", + "\n", + "- `content`: The Cloud Storage path to the file with the text item.\n", + "- `mime_type`: The content type. In our example, it is an `text` file.\n", + "\n", + "For example:\n", + "\n", + " {'content': '[your-bucket]/file1.txt', 'mime_type': 'text'}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_file:automl,text" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "import tensorflow as tf\n", + "\n", + "gcs_test_item_1 = BUCKET_NAME + \"/test1.txt\"\n", + "with tf.io.gfile.GFile(gcs_test_item_1, \"w\") as f:\n", + " f.write(test_item_1 + \"\\n\")\n", + "gcs_test_item_2 = BUCKET_NAME + \"/test2.txt\"\n", + "with tf.io.gfile.GFile(gcs_test_item_2, \"w\") as f:\n", + " f.write(test_item_2 + \"\\n\")\n", + "\n", + "gcs_input_uri = BUCKET_NAME + \"/test.jsonl\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " data = {\"content\": gcs_test_item_1, \"mime_type\": \"text/plain\"}\n", + " f.write(json.dumps(data) + \"\\n\")\n", + " data = {\"content\": gcs_test_item_2, \"mime_type\": \"text/plain\"}\n", + " f.write(json.dumps(data) + \"\\n\")\n", + "\n", + "print(gcs_input_uri)\n", + "! gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your batch prediction requests:\n", + "\n", + "- Single Instance: The batch prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The batch prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and batch prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The batch prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_request:automl,tst" + }, + "source": [ + "### Make batch prediction request\n", + "\n", + "Now that your batch of two test items is ready, let's do the batch request. Use this helper function `create_batch_prediction_job`, with the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the prediction job.\n", + "- `model_name`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `gcs_source_uri`: The Cloud Storage path to the input file -- which you created above.\n", + "- `gcs_destination_output_uri_prefix`: The Cloud Storage path that the service will write the predictions to.\n", + "- `parameters`: Additional filtering parameters for serving prediction results.\n", + "\n", + "The helper function calls the job client service's `create_batch_prediction_job` metho, with the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for Dataset, Model and Pipeline resources.\n", + "- `batch_prediction_job`: The specification for the batch prediction job.\n", + "\n", + "Let's now dive into the specification for the `batch_prediction_job`:\n", + "\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `dedicated_resources`: The compute resources to provision for the batch prediction job.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `starting_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "- `model_parameters`: Additional filtering parameters for serving prediction results. *Note*, text models do not support additional parameters.\n", + "- `input_config`: The input source and format type for the instances to predict.\n", + " - `instances_format`: The format of the batch prediction request file: `jsonl` only supported.\n", + " - `gcs_source`: A list of one or more Cloud Storage paths to your batch prediction requests.\n", + "- `output_config`: The output destination and format for the predictions.\n", + " - `prediction_format`: The format of the batch prediction response file: `jsonl` only supported.\n", + " - `gcs_destination`: The output destination for the predictions.\n", + "\n", + "This call is an asychronous operation. You will print from the response object a few select fields, including:\n", + "\n", + "- `name`: The Vertex fully qualified identifier assigned to the batch prediction job.\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the Model resource.\n", + "- `generate_explanations`: Whether True/False explanations were provided with the predictions (explainability).\n", + "- `state`: The state of the prediction job (pending, running, etc).\n", + "\n", + "Since this call will take a few moments to execute, you will likely get `JobState.JOB_STATE_PENDING` for `state`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_request:automl,tst" + }, + "outputs": [], + "source": [ + "BATCH_MODEL = \"claritin_batch-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_batch_prediction_job(\n", + " display_name,\n", + " model_name,\n", + " gcs_source_uri,\n", + " gcs_destination_output_uri_prefix,\n", + " parameters=None,\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " batch_prediction_job = {\n", + " \"display_name\": display_name,\n", + " # Format: 'projects/{project}/locations/{location}/models/{model_id}'\n", + " \"model\": model_name,\n", + " \"model_parameters\": json_format.ParseDict(parameters, Value()),\n", + " \"input_config\": {\n", + " \"instances_format\": IN_FORMAT,\n", + " \"gcs_source\": {\"uris\": [gcs_source_uri]},\n", + " },\n", + " \"output_config\": {\n", + " \"predictions_format\": OUT_FORMAT,\n", + " \"gcs_destination\": {\"output_uri_prefix\": gcs_destination_output_uri_prefix},\n", + " },\n", + " \"dedicated_resources\": {\n", + " \"machine_spec\": machine_spec,\n", + " \"starting_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " },\n", + " }\n", + " response = clients[\"job\"].create_batch_prediction_job(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + " )\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try:\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", response.labels)\n", + " return response\n", + "\n", + "\n", + "IN_FORMAT = \"jsonl\"\n", + "OUT_FORMAT = \"jsonl\" # [jsonl]\n", + "\n", + "response = create_batch_prediction_job(\n", + " BATCH_MODEL, model_to_deploy_id, gcs_input_uri, BUCKET_NAME, None\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batch_job_id:response" + }, + "source": [ + "Now get the unique identifier for the batch prediction job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batch_job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the batch job\n", + "batch_job_id = response.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split(\"/\")[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_batch_prediction_job" + }, + "source": [ + "### Get information on a batch prediction job\n", + "\n", + "Use this helper function `get_batch_prediction_job`, with the following paramter:\n", + "\n", + "- `job_name`: The Vertex fully qualified identifier for the batch prediction job.\n", + "\n", + "The helper function calls the job client service's `get_batch_prediction_job` method, with the following paramter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the batch prediction job. In this tutorial, you will pass it the Vertex fully qualified identifier for your batch prediction job -- `batch_job_id`\n", + "\n", + "The helper function will return the Cloud Storage path to where the predictions are stored -- `gcs_destination`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_batch_prediction_job" + }, + "outputs": [], + "source": [ + "def get_batch_prediction_job(job_name, silent=False):\n", + " response = clients[\"job\"].get_batch_prediction_job(name=job_name)\n", + " if silent:\n", + " return response.output_config.gcs_destination.output_uri_prefix, response.state\n", + "\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try: # not all data types support explanations\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" error:\", response.error)\n", + " gcs_destination = response.output_config.gcs_destination\n", + " print(\" gcs_destination\")\n", + " print(\" output_uri_prefix:\", gcs_destination.output_uri_prefix)\n", + " return gcs_destination.output_uri_prefix, response.state\n", + "\n", + "\n", + "predictions, state = get_batch_prediction_job(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_the_predictions:automl,tst" + }, + "source": [ + "### Get the predictions\n", + "\n", + "When the batch prediction is done processing, the job state will be `JOB_STATE_SUCCEEDED`.\n", + "\n", + "Finally you view the predictions stored at the Cloud Storage path you set as output. The predictions will be in a JSONL format, which you indicated at the time you made the batch prediction job, under a subfolder starting with the name `prediction`, and under that folder will be a file called `predictions*.jsonl`.\n", + "\n", + "Now display (cat) the contents. You will see multiple JSON objects, one for each prediction.\n", + "\n", + "The first field `text_snippet` is the text file you did the prediction on, and the second field `annotations` is the prediction, which is further broken down into:\n", + "\n", + "- `sentiment`: The predicted sentiment level." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_the_predictions:automl,text" + }, + "outputs": [], + "source": [ + "def get_latest_predictions(gcs_out_dir):\n", + " \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n", + " folders = !gsutil ls $gcs_out_dir\n", + " latest = \"\"\n", + " for folder in folders:\n", + " subfolder = folder.split(\"/\")[-2]\n", + " if subfolder.startswith(\"prediction-\"):\n", + " if subfolder > latest:\n", + " latest = folder[:-1]\n", + " return latest\n", + "\n", + "\n", + "while True:\n", + " predictions, state = get_batch_prediction_job(batch_job_id, True)\n", + " if state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", state)\n", + " if state == aip.JobState.JOB_STATE_FAILED:\n", + " raise Exception(\"Batch Job Failed\")\n", + " else:\n", + " folder = get_latest_predictions(predictions)\n", + " ! gsutil ls $folder/prediction*.jsonl\n", + "\n", + " ! gsutil cat $folder/prediction*.jsonl\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_automl_text_sentiment_analysis_batch.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/automl/showcase_automl_text_sentiment_analysis_online.ipynb b/notebooks/community/gapic/automl/showcase_automl_text_sentiment_analysis_online.ipynb new file mode 100644 index 000000000..288d3c40d --- /dev/null +++ b/notebooks/community/gapic/automl/showcase_automl_text_sentiment_analysis_online.ipynb @@ -0,0 +1,1553 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: AutoML text sentiment analysis model for online prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to create text sentiment analysis models and do online prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:claritin,tst" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Crowdflower Claritin-Twitter dataset](https://data.world/crowdflower/claritin-twitter) from [data.world Datasets](https://data.world). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,online_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML text sentiment analysis model and deploy for online prediction from a Python script using the Vertex client library. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model`." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Set constants unique to AutoML datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the `Dataset` resource service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the `Dataset` resource service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:tst" + }, + "outputs": [], + "source": [ + "# Text Dataset type\n", + "DATA_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/metadata/text_1.0.0.yaml\"\n", + "# Text Labeling type\n", + "LABEL_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/ioformat/text_sentiment_io_format_1.0.0.yaml\"\n", + "# Text Training task\n", + "TRAINING_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_text_sentiment_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML text sentiment analysis model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:automl,online_prediction" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training.\n", + "- Endpoint Service for deployment.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:automl,online_prediction" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "\n", + "def create_dataset(name, schema, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " dataset = aip.Dataset(\n", + " display_name=name, metadata_schema_uri=schema, labels=labels\n", + " )\n", + "\n", + " operation = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"claritin-\" + TIMESTAMP, DATA_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:text,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for text has a couple of requirements for your text data.\n", + "\n", + "- Text examples must be stored in a CSV or JSONL file." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:tst,u_dataset,csv" + }, + "source": [ + "#### CSV\n", + "\n", + "For text sentiment analysis, the CSV file has a few requirements:\n", + "\n", + "- No heading.\n", + "- First column is the text example or Cloud Storage path to text file.\n", + "- Second column the label (i.e., sentiment).\n", + "- Third column is the maximum sentiment value. For example, if the range is 0 to 3, then the maximum value is 3." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:claritin,csv,tst" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://cloud-samples-data/language/claritin.csv\"\n", + "SENTIMENT_MAX = 4" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:csv" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Crowdflower Claritin-Twitter dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:csv" + }, + "outputs": [], + "source": [ + "if \"IMPORT_FILES\" in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_data" + }, + "source": [ + "### Import data\n", + "\n", + "Now, import the data into your Vertex Dataset resource. Use this helper function `import_data` to import the data. The function does the following:\n", + "\n", + "- Uses the `Dataset` client.\n", + "- Calls the client method `import_data`, with the following parameters:\n", + " - `name`: The human readable name you give to the `Dataset` resource (e.g., claritin).\n", + " - `import_configs`: The import configuration.\n", + "\n", + "- `import_configs`: A Python list containing a dictionary, with the key/value entries:\n", + " - `gcs_sources`: A list of URIs to the paths of the one or more index files.\n", + " - `import_schema_uri`: The schema identifying the labeling type.\n", + "\n", + "The `import_data()` method returns a long running `operation` object. This will take a few minutes to complete. If you are in a live tutorial, this would be a good time to ask questions, or take a personal break." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_data" + }, + "outputs": [], + "source": [ + "def import_data(dataset, gcs_sources, schema):\n", + " config = [{\"gcs_source\": {\"uris\": gcs_sources}, \"import_schema_uri\": schema}]\n", + " print(\"dataset:\", dataset_id)\n", + " start_time = time.time()\n", + " try:\n", + " operation = clients[\"dataset\"].import_data(\n", + " name=dataset_id, import_configs=config\n", + " )\n", + " print(\"Long running operation:\", operation.operation.name)\n", + "\n", + " result = operation.result()\n", + " print(\"result:\", result)\n", + " print(\"time:\", int(time.time() - start_time), \"secs\")\n", + " print(\"error:\", operation.exception())\n", + " print(\"meta :\", operation.metadata)\n", + " print(\n", + " \"after: running:\",\n", + " operation.running(),\n", + " \"done:\",\n", + " operation.done(),\n", + " \"cancelled:\",\n", + " operation.cancelled(),\n", + " )\n", + "\n", + " return operation\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "import_data(dataset_id, [IMPORT_FILE], LABEL_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML text sentiment analysis model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_pipeline:automl" + }, + "source": [ + "### Create a training pipeline\n", + "\n", + "You may ask, what do we use a pipeline for? You typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `pipeline_name`: A human readable name for the pipeline job.\n", + "- `model_name`: A human readable name for the model.\n", + "- `dataset`: The Vertex fully qualified dataset identifier.\n", + "- `schema`: The dataset labeling (annotation) training schema.\n", + "- `task`: A dictionary describing the requirements for the training job.\n", + "\n", + "The helper function calls the `Pipeline` client service'smethod `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "Let's look now deeper into the *minimal* requirements for constructing a `training_pipeline` specification:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The dataset labeling (annotation) training schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A human readable name for the model.\n", + "- `input_data_config`: The dataset specification.\n", + " - `dataset_id`: The Vertex dataset identifier only (non-fully qualified) -- this is the last part of the fully-qualified identifier.\n", + " - `fraction_split`: If specified, the percentages of the dataset to use for training, test and validation. Otherwise, the percentages are automatically selected by AutoML." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_pipeline:automl" + }, + "outputs": [], + "source": [ + "def create_pipeline(pipeline_name, model_name, dataset, schema, task):\n", + "\n", + " dataset_id = dataset.split(\"/\")[-1]\n", + "\n", + " input_config = {\n", + " \"dataset_id\": dataset_id,\n", + " \"fraction_split\": {\n", + " \"training_fraction\": 0.8,\n", + " \"validation_fraction\": 0.1,\n", + " \"test_fraction\": 0.1,\n", + " },\n", + " }\n", + "\n", + " training_pipeline = {\n", + " \"display_name\": pipeline_name,\n", + " \"training_task_definition\": schema,\n", + " \"training_task_inputs\": task,\n", + " \"input_data_config\": input_config,\n", + " \"model_to_upload\": {\"display_name\": model_name},\n", + " }\n", + "\n", + " try:\n", + " pipeline = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " )\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "task_requirements:automl,tst" + }, + "source": [ + "### Construct the task requirements\n", + "\n", + "Next, construct the task requirements. Unlike other parameters which take a Python (JSON-like) dictionary, the `task` field takes a Google protobuf Struct, which is very similar to a Python dictionary. Use the `json_format.ParseDict` method for the conversion.\n", + "\n", + "The minimal fields we need to specify are:\n", + "\n", + "- `sentiment_max`: The maximum value for the sentiment (e.g., 4).\n", + "\n", + "Finally, create the pipeline by calling the helper function `create_pipeline`, which returns an instance of a training pipeline object." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_requirements:automl,tst" + }, + "outputs": [], + "source": [ + "PIPE_NAME = \"claritin_pipe-\" + TIMESTAMP\n", + "MODEL_NAME = \"claritin_model-\" + TIMESTAMP\n", + "\n", + "task = json_format.ParseDict(\n", + " {\n", + " \"sentiment_max\": SENTIMENT_MAX,\n", + " },\n", + " Value(),\n", + ")\n", + "\n", + "response = create_pipeline(PIPE_NAME, MODEL_NAME, dataset_id, TRAINING_SCHEMA, task)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 180 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "model_information" + }, + "source": [ + "## Model information\n", + "\n", + "Now that your model is trained, you can get some information on your model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:automl" + }, + "source": [ + "## Evaluate the Model resource\n", + "\n", + "Now find out how good the model service believes your model is. As part of training, some portion of the dataset was set aside as the test (holdout) data, which is used by the pipeline service to evaluate the model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "list_model_evaluations:automl,tst" + }, + "source": [ + "### List evaluations for all slices\n", + "\n", + "Use this helper function `list_model_evaluations`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified model identifier for the `Model` resource.\n", + "\n", + "This helper function uses the model client service's `list_model_evaluations` method, which takes the same parameter. The response object from the call is a list, where each element is an evaluation metric.\n", + "\n", + "For each evaluation -- you probably only have one, we then print all the key names for each metric in the evaluation, and for a small set (`meanAbsoluteError` and `precision`) you will print the result." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "list_model_evaluations:automl,tst" + }, + "outputs": [], + "source": [ + "def list_model_evaluations(name):\n", + " response = clients[\"model\"].list_model_evaluations(parent=name)\n", + " for evaluation in response:\n", + " print(\"model_evaluation\")\n", + " print(\" name:\", evaluation.name)\n", + " print(\" metrics_schema_uri:\", evaluation.metrics_schema_uri)\n", + " metrics = json_format.MessageToDict(evaluation._pb.metrics)\n", + " for metric in metrics.keys():\n", + " print(metric)\n", + " print(\"meanAbsoluteError\", metrics[\"meanAbsoluteError\"])\n", + " print(\"precision\", metrics[\"precision\"])\n", + "\n", + " return evaluation.name\n", + "\n", + "\n", + "last_evaluation = list_model_evaluations(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:automl" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created with AutoML. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"claritin_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:automatic" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `automatic_resources`: This refers to how many redundant compute instances (replicas). For this example, we set it to one (no replication).\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:automatic" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"claritin_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"automatic_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " },\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_item" + }, + "source": [ + "### Get test item\n", + "\n", + "You will use an arbitrary example out of the dataset as a test item. Don't be concerned that the example was likely used in training the model -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:automl,tst,csv" + }, + "outputs": [], + "source": [ + "test_item = ! gsutil cat $IMPORT_FILE | head -n1\n", + "if len(test_item[0]) == 3:\n", + " _, test_item, test_label, max = str(test_item[0]).split(\",\")\n", + "else:\n", + " test_item, test_label, max = str(test_item[0]).split(\",\")\n", + "\n", + "print(test_item, test_label)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "predict_item:automl,tst" + }, + "source": [ + "### Make a prediction\n", + "\n", + "Now you have a test item. Use this helper function `predict_item`, which takes the following parameters:\n", + "\n", + "- `filename`: The Cloud Storage path to the test item.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `parameters_dict`: Additional filtering parameters for serving prediction results.\n", + "\n", + "This function calls the prediction client service's `predict` method with the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `instances`: A list of instances (text files) to predict.\n", + "- `parameters`: Additional filtering parameters for serving prediction results. *Note*, text models do not support additional parameters.\n", + "\n", + "#### Request\n", + "\n", + "The format of each instance is:\n", + "\n", + " { 'content': text_item }\n", + "\n", + "Since the `predict()` method can take multiple items (instances), you send your single test item as a list of one test item. As a final step, you package the instances list into Google's protobuf format -- which is what you pass to the `predict()` method.\n", + "\n", + "#### Response\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding text in the request. You will see in the output for each prediction -- in our case there is just one:\n", + "\n", + "- The sentiment rating" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "predict_item:automl,tst" + }, + "outputs": [], + "source": [ + "def predict_item(data, endpoint, parameters_dict):\n", + "\n", + " parameters = json_format.ParseDict(parameters_dict, Value())\n", + "\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [{\"content\": data}]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", dict(prediction))\n", + " return response\n", + "\n", + "\n", + "response = predict_item(test_item, endpoint_id, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_automl_text_sentiment_analysis_online.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/automl/showcase_automl_video_action_recognition_batch.ipynb b/notebooks/community/gapic/automl/showcase_automl_video_action_recognition_batch.ipynb new file mode 100644 index 000000000..e7fab13d9 --- /dev/null +++ b/notebooks/community/gapic/automl/showcase_automl_video_action_recognition_batch.ipynb @@ -0,0 +1,1771 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: AutoML video action recognition model for batch prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to create video action recognition models and do batch prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:golf,var" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the golf swing recognition portion of the [Human Motion dataset](https://todo) from [MIT](http://cbcl.mit.edu/publications/ps/Kuehne_etal_iccv11.pdf). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model will predict the start frame where a golf swing begins." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,batch_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML video action recognition model from a Python script, and then do a batch prediction using the Vertex client library. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Make a batch prediction.\n", + "\n", + "There is one key difference between using batch prediction and using online prediction:\n", + "\n", + "* Prediction Service: Does an on-demand prediction for the entire set of instances (i.e., one or more data items) and returns the results in real-time.\n", + "\n", + "* Batch Prediction Service: Does a queued (batch) prediction for the entire set of instances in the background and stores the results in a Cloud Storage bucket when ready." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = '--user'\n", + "else:\n", + " USER_FLAG = ''\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" #@param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = 'us-central1' #@param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" #@param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Set constants unique to AutoML datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the `Dataset` resource service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the `Dataset` resource service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:var" + }, + "outputs": [], + "source": [ + "# Video Dataset type\n", + "DATA_SCHEMA = 'gs://google-cloud-aiplatform/schema/dataset/metadata/video_1.0.0.yaml'\n", + "# Video Labeling type\n", + "LABEL_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/ioformat/video_action_recognition_io_format_1.0.0.yaml\"\n", + "# Video Training task\n", + "TRAINING_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_video_action_recognition_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:prediction" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for prediction.\n", + "\n", + "Set the variable `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")))\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:automl" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "For AutoML batch prediction, the container image for the serving binary is pre-determined by the Vertex prediction service. More specifically, the service will pick the appropriate container for the model depending on the hardware accelerator you selected." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for prediction.\n", + "\n", + "- Set the variable `DEPLOY_COMPUTE` to configure the compute resources for the VM you will use for prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = 'n1-standard'\n", + "\n", + "VCPU = '4'\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + '-' + VCPU\n", + "print('Deploy machine type', DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML video action recognition model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:automl,batch_prediction" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training.\n", + "- Job Service for batch prediction and custom training." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:automl,batch_prediction" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(\n", + " client_options=client_options\n", + " )\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(\n", + " client_options=client_options\n", + " )\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(\n", + " client_options=client_options\n", + " )\n", + " return client\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(\n", + " client_options=client_options\n", + " )\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients['dataset'] = create_dataset_client()\n", + "clients['model'] = create_model_client()\n", + "clients['pipeline'] = create_pipeline_client()\n", + "clients['job'] = create_job_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "def create_dataset(name, schema, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " dataset = aip.Dataset(display_name=name, metadata_schema_uri=schema, labels=labels)\n", + "\n", + " operation = clients['dataset'].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"golf-\" + TIMESTAMP, DATA_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split('/')[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:video,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for video has some requirements for your data.\n", + "\n", + "- Videos must be stored in a Cloud Storage bucket.\n", + "- Each video file must be in a video format (MPG, AVI, ...).\n", + "- There must be an index file stored in your Cloud Storage bucket that contains the path and label for each video.\n", + "- The index file must be either CSV or JSONL." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:var,u_dataset,csv" + }, + "source": [ + "#### CSV\n", + "\n", + "For video action recognition, the CSV index file has a few requirements:\n", + "\n", + "- No heading.\n", + "- First column is the Cloud Storage path to the video.\n", + "- Second column is the time offset for the start of the video segment to analyze.\n", + "- Third column is the time offset for the end of the video segment to analyze.\n", + "- Fourth column is label for the action (e.g., swing).\n", + "- Fifth column is the time offset for the recognized action." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:golf,csv,var" + }, + "outputs": [], + "source": [ + "IMPORT_FILES = ['gs://automl-video-demo-data/hmdb_golf_swing_train.csv', 'gs://automl-video-demo-data/hmdb_golf_swing_test.csv']" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:csv" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Golf Swings dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:csv" + }, + "outputs": [], + "source": [ + "if 'IMPORT_FILES' in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_data:files" + }, + "source": [ + "### Import data\n", + "\n", + "Now, import the data into your Vertex Dataset resource. Use this helper function `import_data` to import the data. The function does the following:\n", + "\n", + "- Uses the `Dataset` client.\n", + "- Calls the client method `import_data`, with the following parameters:\n", + " - `name`: The human readable name you give to the `Dataset` resource (e.g., golf).\n", + " - `import_configs`: The import configuration.\n", + "\n", + "- `import_configs`: A Python list containing a dictionary, with the key/value entries:\n", + " - `gcs_sources`: A list of URIs to the paths of the one or more index files.\n", + " - `import_schema_uri`: The schema identifying the labeling type.\n", + "\n", + "The `import_data()` method returns a long running `operation` object. This will take a few minutes to complete. If you are in a live tutorial, this would be a good time to ask questions, or take a personal break." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_data:files" + }, + "outputs": [], + "source": [ + "def import_data(dataset, gcs_sources, schema):\n", + " config = [{\n", + " 'gcs_source': {'uris': gcs_sources},\n", + " 'import_schema_uri': schema\n", + " }]\n", + " print(\"dataset:\", dataset_id)\n", + " start_time = time.time()\n", + " try:\n", + " operation = clients['dataset'].import_data(name=dataset_id, import_configs=config)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + "\n", + " result = operation.result()\n", + " print(\"result:\", result)\n", + " print(\"time:\", int(time.time() - start_time), \"secs\")\n", + " print(\"error:\", operation.exception())\n", + " print(\"meta :\", operation.metadata)\n", + " print(\"after: running:\", operation.running(), \"done:\", operation.done(), \"cancelled:\", operation.cancelled())\n", + "\n", + " return operation\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "import_data(dataset_id, IMPORT_FILES, LABEL_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML video action recognition model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_pipeline:automl,video" + }, + "source": [ + "### Create a training pipeline\n", + "\n", + "You may ask, what do we use a pipeline for? You typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `pipeline_name`: A human readable name for the pipeline job.\n", + "- `model_name`: A human readable name for the model.\n", + "- `dataset`: The Vertex fully qualified dataset identifier.\n", + "- `schema`: The dataset labeling (annotation) training schema.\n", + "- `task`: A dictionary describing the requirements for the training job.\n", + "\n", + "The helper function calls the `Pipeline` client service'smethod `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "Let's look now deeper into the *minimal* requirements for constructing a `training_pipeline` specification:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The dataset labeling (annotation) training schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A human readable name for the model.\n", + "- `input_data_config`: The dataset specification.\n", + " - `dataset_id`: The Vertex dataset identifier only (non-fully qualified) -- this is the last part of the fully-qualified identifier.\n", + " - `fraction_split`: If specified, the percentages of the dataset to use for training, test and validation. Otherwise, the percentages are automatically selected by AutoML.\n", + " - Note for video, validation split is not supported -- only training and test." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_pipeline:automl,video" + }, + "outputs": [], + "source": [ + " def create_pipeline(pipeline_name, model_name, dataset, schema, task):\n", + "\n", + " dataset_id = dataset.split('/')[-1]\n", + "\n", + " input_config = {'dataset_id': dataset_id,\n", + " 'fraction_split': {\n", + " 'training_fraction': 0.8,\n", + " 'test_fraction': 0.2\n", + " }}\n", + "\n", + " training_pipeline = {\n", + " \"display_name\": pipeline_name,\n", + " \"training_task_definition\": schema,\n", + " \"training_task_inputs\": task,\n", + " \"input_data_config\": input_config,\n", + " \"model_to_upload\": {\"display_name\": model_name},\n", + " }\n", + "\n", + " try:\n", + " pipeline = clients['pipeline'].create_training_pipeline(parent=PARENT, training_pipeline=training_pipeline)\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "task_requirements:automl,var" + }, + "source": [ + "### Construct the task requirements\n", + "\n", + "Next, construct the task requirements. Unlike other parameters which take a Python (JSON-like) dictionary, the `task` field takes a Google protobuf Struct, which is very similar to a Python dictionary. Use the `json_format.ParseDict` method for the conversion.\n", + "\n", + "The minimal fields you need to specify are:\n", + "\n", + "- `model_type`: The type of deployed model, ex. CLOUD for deploying to Google Cloud.\n", + "\n", + "Finally, create the pipeline by calling the helper function `create_pipeline`, which returns an instance of a training pipeline object." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_requirements:automl,var" + }, + "outputs": [], + "source": [ + "PIPE_NAME = \"golf_pipe-\" + TIMESTAMP\n", + "MODEL_NAME = \"golf_model-\" + TIMESTAMP\n", + "\n", + "task = json_format.ParseDict({'model_type': \"CLOUD\",\n", + " }, Value())\n", + "\n", + "response = create_pipeline(PIPE_NAME, MODEL_NAME, dataset_id, TRAINING_SCHEMA, task)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split('/')[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients['pipeline'].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 240 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "model_information" + }, + "source": [ + "## Model information\n", + "\n", + "Now that your model is trained, you can get some information on your model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:automl" + }, + "source": [ + "## Evaluate the Model resource\n", + "\n", + "Now find out how good the model service believes your model is. As part of training, some portion of the dataset was set aside as the test (holdout) data, which is used by the pipeline service to evaluate the model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "list_model_evaluations:automl,var" + }, + "source": [ + "### List evaluations for all slices\n", + "\n", + "Use this helper function `list_model_evaluations`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified model identifier for the `Model` resource.\n", + "\n", + "This helper function uses the model client service's `list_model_evaluations` method, which takes the same parameter. The response object from the call is a list, where each element is an evaluation metric.\n", + "\n", + "For each evaluation -- you probably only have one, you then print all the key names for each metric in the evaluation, and for a small set (`videoActionMetrics`) you will print the result." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "list_model_evaluations:automl,var" + }, + "outputs": [], + "source": [ + "def list_model_evaluations(name):\n", + " response = clients['model'].list_model_evaluations(parent=name)\n", + " for evaluation in response:\n", + " print(\"model_evaluation\")\n", + " print(\" name:\", evaluation.name)\n", + " print(\" metrics_schema_uri:\", evaluation.metrics_schema_uri)\n", + " metrics = json_format.MessageToDict(evaluation._pb.metrics)\n", + " for metric in metrics.keys():\n", + " print(metric)\n", + " print('videoActionMetrics', metrics['videoActionMetrics'])\n", + "\n", + "\n", + " return evaluation.name\n", + "\n", + "\n", + "last_evaluation = list_model_evaluations(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy:batch_prediction" + }, + "source": [ + "## Model deployment for batch prediction\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created for batch prediction. This differs from deploying a `Model` resource for on-demand prediction.\n", + "\n", + "For online prediction, you:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource.\n", + "\n", + "3. Make online prediction requests to the `Endpoint` resource.\n", + "\n", + "For batch-prediction, you:\n", + "\n", + "1. Create a batch prediction job.\n", + "\n", + "2. The job service will provision resources for the batch prediction request.\n", + "\n", + "3. The results of the batch prediction request are returned to the caller.\n", + "\n", + "4. The job service will unprovision the resoures for the batch prediction request." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a batch prediction request\n", + "\n", + "Now do a batch prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_items:batch_prediction" + }, + "source": [ + "### Get test item(s)\n", + "\n", + "Now do a batch prediction to your Vertex model. You will use arbitrary examples out of the dataset as a test items. Don't be concerned that the examples were likely used in training the model -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_items:automl,var,csv" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "import_file = IMPORT_FILES[0]\n", + "test_items = ! gsutil cat $import_file | head -n2\n", + "\n", + "cols = str(test_items[0]).split(',')\n", + "test_item_1 = str(cols[0])\n", + "test_label_1 = str(cols[-1])\n", + "\n", + "cols = str(test_items[1]).split(',')\n", + "test_item_2 = str(cols[0])\n", + "test_label_2 = str(cols[-1])\n", + "\n", + "print(test_item_1, test_label_1)\n", + "print(test_item_2, test_label_2)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:automl,video" + }, + "source": [ + "### Make a batch input file\n", + "\n", + "Now make a batch input file, which you store in your local Cloud Storage bucket. The batch input file can be either CSV or JSONL. You will use JSONL in this tutorial. For JSONL file, you make one dictionary entry per line for each video. The dictionary contains the key/value pairs:\n", + "\n", + "- `content`: The Cloud Storage path to the video.\n", + "- `mimeType`: The content type. In our example, it is an `avi` file.\n", + "- `timeSegmentStart`: The start timestamp in the video to do prediction on. *Note*, the timestamp must be specified as a string and followed by s (second), m (minute) or h (hour).\n", + "- `timeSegmentEnd`: The end timestamp in the video to do prediction on." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_file:automl,video" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "import tensorflow as tf\n", + "\n", + "gcs_input_uri = BUCKET_NAME + '/test.jsonl'\n", + "with tf.io.gfile.GFile(gcs_input_uri, 'w') as f:\n", + " data = { \"content\": test_item_1, \"mimeType\": \"video/avi\", \"timeSegmentStart\": \"0.0s\", 'timeSegmentEnd': '5.0s' }\n", + " f.write(json.dumps(data) + '\\n')\n", + " data = { \"content\": test_item_2, \"mimeType\": \"video/avi\", \"timeSegmentStart\": \"0.0s\", 'timeSegmentEnd': '5.0s' }\n", + " f.write(json.dumps(data) + '\\n')\n", + "\n", + "print(gcs_input_uri)\n", + "! gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your batch prediction requests:\n", + "\n", + "- Single Instance: The batch prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The batch prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and batch prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The batch prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_request:automl,var" + }, + "source": [ + "### Make batch prediction request\n", + "\n", + "Now that your batch of two test items is ready, let's do the batch request. Use this helper function `create_batch_prediction_job`, with the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the prediction job.\n", + "- `model_name`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `gcs_source_uri`: The Cloud Storage path to the input file -- which you created above.\n", + "- `gcs_destination_output_uri_prefix`: The Cloud Storage path that the service will write the predictions to.\n", + "- `parameters`: Additional filtering parameters for serving prediction results.\n", + "\n", + "The helper function calls the job client service's `create_batch_prediction_job` metho, with the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for Dataset, Model and Pipeline resources.\n", + "- `batch_prediction_job`: The specification for the batch prediction job.\n", + "\n", + "Let's now dive into the specification for the `batch_prediction_job`:\n", + "\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `dedicated_resources`: The compute resources to provision for the batch prediction job.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `starting_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "- `model_parameters`: Additional filtering parameters for serving prediction results.\n", + " - `confidenceThreshold`: The minimum confidence threshold on doing a prediction.\n", + " - `maxPredictions`: The maximum number of predictions to return per action, sorted by confidence.\n", + "- `input_config`: The input source and format type for the instances to predict.\n", + " - `instances_format`: The format of the batch prediction request file: `csv` or `jsonl`.\n", + " - `gcs_source`: A list of one or more Cloud Storage paths to your batch prediction requests.\n", + "- `output_config`: The output destination and format for the predictions.\n", + " - `prediction_format`: The format of the batch prediction response file: `jsonl` only.\n", + " - `gcs_destination`: The output destination for the predictions.\n", + "\n", + "You might ask, how does confidence_threshold affect the model accuracy? The threshold won't change the accuracy. What it changes is recall and precision.\n", + "\n", + " - Precision: The higher the precision the more likely what is predicted is the correct prediction, but return fewer predictions. Increasing the confidence threshold increases precision.\n", + " - Recall: The higher the recall the more likely a correct prediction is returned in the result, but return more prediction with incorrect prediction. Decreasing the confidence threshold increases recall.\n", + "\n", + "In this example, you will predict for precision. You set the confidence threshold to 0.5 and the maximum number of predictions for an action to two. Since, all the confidence values across the classes must add up to one, there are only two possible outcomes:\n", + "\n", + " 1. There is a tie, both 0.5, and returns two predictions.\n", + " 2. One value is above 0.5 and the rest are below 0.5, and returns one prediction.\n", + "\n", + "This call is an asychronous operation. You will print from the response object a few select fields, including:\n", + "\n", + "- `name`: The Vertex fully qualified identifier assigned to the batch prediction job.\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the Model resource.\n", + "- `generate_explanations`: Whether True/False explanations were provided with the predictions (explainability).\n", + "- `state`: The state of the prediction job (pending, running, etc).\n", + "\n", + "Since this call will take a few moments to execute, you will likely get `JobState.JOB_STATE_PENDING` for `state`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_request:automl,var" + }, + "outputs": [], + "source": [ + "BATCH_MODEL = \"golf_batch-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_batch_prediction_job(display_name, model_name, gcs_source_uri, gcs_destination_output_uri_prefix, parameters=None):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " batch_prediction_job = {\n", + " \"display_name\": display_name,\n", + " # Format: 'projects/{project}/locations/{location}/models/{model_id}'\n", + " \"model\": model_name,\n", + " \"model_parameters\": json_format.ParseDict(parameters, Value()),\n", + " \"input_config\": {\n", + " \"instances_format\": IN_FORMAT,\n", + " \"gcs_source\": {\"uris\": [gcs_source_uri]},\n", + " },\n", + " \"output_config\": {\n", + " \"predictions_format\": OUT_FORMAT,\n", + " \"gcs_destination\": {\"output_uri_prefix\": gcs_destination_output_uri_prefix},\n", + " },\n", + " \"dedicated_resources\": {\n", + " \"machine_spec\": machine_spec,\n", + " \"starting_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES\n", + " }\n", + "\n", + " }\n", + " response = clients['job'].create_batch_prediction_job(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + " )\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try:\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", response.labels)\n", + " return response\n", + "\n", + "\n", + "IN_FORMAT = 'jsonl'\n", + "OUT_FORMAT = 'jsonl' # [jsonl]\n", + "\n", + "response = create_batch_prediction_job(BATCH_MODEL, model_to_deploy_id, gcs_input_uri, BUCKET_NAME,\n", + " {'confidenceThreshold': 0.5, 'maxPredictions': 2})" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batch_job_id:response" + }, + "source": [ + "Now get the unique identifier for the batch prediction job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batch_job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the batch job\n", + "batch_job_id = response.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split('/')[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_batch_prediction_job" + }, + "source": [ + "### Get information on a batch prediction job\n", + "\n", + "Use this helper function `get_batch_prediction_job`, with the following paramter:\n", + "\n", + "- `job_name`: The Vertex fully qualified identifier for the batch prediction job.\n", + "\n", + "The helper function calls the job client service's `get_batch_prediction_job` method, with the following paramter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the batch prediction job. In this tutorial, you will pass it the Vertex fully qualified identifier for your batch prediction job -- `batch_job_id`\n", + "\n", + "The helper function will return the Cloud Storage path to where the predictions are stored -- `gcs_destination`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_batch_prediction_job" + }, + "outputs": [], + "source": [ + "def get_batch_prediction_job(job_name, silent=False):\n", + " response = clients['job'].get_batch_prediction_job(name=job_name)\n", + " if silent:\n", + " return response.output_config.gcs_destination.output_uri_prefix, response.state\n", + "\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try: # not all data types support explanations\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" error:\", response.error)\n", + " gcs_destination = response.output_config.gcs_destination\n", + " print(\" gcs_destination\")\n", + " print(\" output_uri_prefix:\", gcs_destination.output_uri_prefix)\n", + " return gcs_destination.output_uri_prefix, response.state\n", + "\n", + "\n", + "predictions, state = get_batch_prediction_job(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_the_predictions:automl,var" + }, + "source": [ + "### Get the predictions\n", + "\n", + "When the batch prediction is done processing, the job state will be `JOB_STATE_SUCCEEDED`.\n", + "\n", + "Finally you view the predictions stored at the Cloud Storage path you set as output. The predictions will be in a JSONL format, which you indicated at the time you made the batch prediction job, under a subfolder starting with the name `prediction`, and under that folder will be a file called `predictions*.jsonl`.\n", + "\n", + "Now display (cat) the contents. You will see multiple JSON objects, one for each prediction.\n", + "\n", + "For each prediction:\n", + "\n", + "- `content`: The video that was input for the prediction request.\n", + "- `displayName`: The prediction action.\n", + "- `confidence`: The confidence in the prediction between 0 and 1.\n", + "- `timeSegmentStart/timeSegmentEnd`: The time offset of the start and end of the predicted action." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_the_predictions:automl,video" + }, + "outputs": [], + "source": [ + "def get_latest_predictions(gcs_out_dir):\n", + " ''' Get the latest prediction subfolder using the timestamp in the subfolder name'''\n", + " folders = !gsutil ls $gcs_out_dir\n", + " latest = \"\"\n", + " for folder in folders:\n", + " subfolder = folder.split('/')[-2]\n", + " if subfolder.startswith('prediction-'):\n", + " if subfolder > latest:\n", + " latest = folder[:-1]\n", + " return latest\n", + "\n", + "\n", + "while True:\n", + " predictions, state = get_batch_prediction_job(batch_job_id, True)\n", + " if state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", state)\n", + " if state == aip.JobState.JOB_STATE_FAILED:\n", + " raise Exception(\"Batch Job Failed\")\n", + " else:\n", + " folder = get_latest_predictions(predictions)\n", + " ! gsutil ls $folder/prediction*.jsonl\n", + "\n", + " ! gsutil cat $folder/prediction*.jsonl\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and 'dataset_id' in globals():\n", + " clients['dataset'].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and 'pipeline_id' in globals():\n", + " clients['pipeline'].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and 'model_to_deploy_id' in globals():\n", + " clients['model'].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and 'endpoint_id' in globals():\n", + " clients['endpoint'].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and 'batch_job_id' in globals():\n", + " clients['job'].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and 'job_id' in globals():\n", + " clients['job'].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and 'hpt_job_id' in globals():\n", + " clients['job'].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and 'BUCKET_NAME' in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_automl_video_action_recognition_batch.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/automl/showcase_automl_video_classification_batch.ipynb b/notebooks/community/gapic/automl/showcase_automl_video_classification_batch.ipynb new file mode 100644 index 000000000..f987d0395 --- /dev/null +++ b/notebooks/community/gapic/automl/showcase_automl_video_classification_batch.ipynb @@ -0,0 +1,1764 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: AutoML video classification model for batch prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to create video classification models and do batch prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:hmdb,vcn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Human Motion dataset](https://TODO) from [MIT](http://cbcl.mit.edu/publications/ps/Kuehne_etal_iccv11.pdf). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,batch_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML video classification model from a Python script, and then do a batch prediction using the Vertex client library. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Make a batch prediction.\n", + "\n", + "There is one key difference between using batch prediction and using online prediction:\n", + "\n", + "* Prediction Service: Does an on-demand prediction for the entire set of instances (i.e., one or more data items) and returns the results in real-time.\n", + "\n", + "* Batch Prediction Service: Does a queued (batch) prediction for the entire set of instances in the background and stores the results in a Cloud Storage bucket when ready." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = '--user'\n", + "else:\n", + " USER_FLAG = ''\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" #@param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = 'us-central1' #@param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" #@param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Set constants unique to AutoML datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the `Dataset` resource service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the `Dataset` resource service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:vcn" + }, + "outputs": [], + "source": [ + "# Video Dataset type\n", + "DATA_SCHEMA = 'gs://google-cloud-aiplatform/schema/dataset/metadata/video_1.0.0.yaml'\n", + "# Video Labeling type\n", + "LABEL_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/ioformat/video_classification_io_format_1.0.0.yaml\"\n", + "# Video Training task\n", + "TRAINING_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_video_classification_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:prediction" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for prediction.\n", + "\n", + "Set the variable `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")))\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:automl" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "For AutoML batch prediction, the container image for the serving binary is pre-determined by the Vertex prediction service. More specifically, the service will pick the appropriate container for the model depending on the hardware accelerator you selected." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for prediction.\n", + "\n", + "- Set the variable `DEPLOY_COMPUTE` to configure the compute resources for the VM you will use for prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = 'n1-standard'\n", + "\n", + "VCPU = '4'\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + '-' + VCPU\n", + "print('Deploy machine type', DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML video classification model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:automl,batch_prediction" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training.\n", + "- Job Service for batch prediction and custom training." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:automl,batch_prediction" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(\n", + " client_options=client_options\n", + " )\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(\n", + " client_options=client_options\n", + " )\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(\n", + " client_options=client_options\n", + " )\n", + " return client\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(\n", + " client_options=client_options\n", + " )\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients['dataset'] = create_dataset_client()\n", + "clients['model'] = create_model_client()\n", + "clients['pipeline'] = create_pipeline_client()\n", + "clients['job'] = create_job_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "def create_dataset(name, schema, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " dataset = aip.Dataset(display_name=name, metadata_schema_uri=schema, labels=labels)\n", + "\n", + " operation = clients['dataset'].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"hmdb,tst-\" + TIMESTAMP, DATA_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split('/')[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:video,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for video has some requirements for your data.\n", + "\n", + "- Videos must be stored in a Cloud Storage bucket.\n", + "- Each video file must be in a video format (MPG, AVI, ...).\n", + "- There must be an index file stored in your Cloud Storage bucket that contains the path and label for each video.\n", + "- The index file must be either CSV or JSONL." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:vcn,u_dataset,csv" + }, + "source": [ + "#### CSV\n", + "\n", + "For video classification, the CSV index file has a few requirements:\n", + "\n", + "- No heading.\n", + "- First column is the Cloud Storage path to the video.\n", + "- Second column is the label." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:hmdb,csv,vcn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = 'gs://automl-video-demo-data/hmdb_split1_5classes_train_inf.csv'" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:csv" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the MIT Human Motion dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:csv" + }, + "outputs": [], + "source": [ + "if 'IMPORT_FILES' in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_data" + }, + "source": [ + "### Import data\n", + "\n", + "Now, import the data into your Vertex Dataset resource. Use this helper function `import_data` to import the data. The function does the following:\n", + "\n", + "- Uses the `Dataset` client.\n", + "- Calls the client method `import_data`, with the following parameters:\n", + " - `name`: The human readable name you give to the `Dataset` resource (e.g., hmdb,tst).\n", + " - `import_configs`: The import configuration.\n", + "\n", + "- `import_configs`: A Python list containing a dictionary, with the key/value entries:\n", + " - `gcs_sources`: A list of URIs to the paths of the one or more index files.\n", + " - `import_schema_uri`: The schema identifying the labeling type.\n", + "\n", + "The `import_data()` method returns a long running `operation` object. This will take a few minutes to complete. If you are in a live tutorial, this would be a good time to ask questions, or take a personal break." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_data" + }, + "outputs": [], + "source": [ + "def import_data(dataset, gcs_sources, schema):\n", + " config = [{\n", + " 'gcs_source': {'uris': gcs_sources},\n", + " 'import_schema_uri': schema\n", + " }]\n", + " print(\"dataset:\", dataset_id)\n", + " start_time = time.time()\n", + " try:\n", + " operation = clients['dataset'].import_data(name=dataset_id, import_configs=config)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + "\n", + " result = operation.result()\n", + " print(\"result:\", result)\n", + " print(\"time:\", int(time.time() - start_time), \"secs\")\n", + " print(\"error:\", operation.exception())\n", + " print(\"meta :\", operation.metadata)\n", + " print(\"after: running:\", operation.running(), \"done:\", operation.done(), \"cancelled:\", operation.cancelled())\n", + "\n", + " return operation\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "import_data(dataset_id, [IMPORT_FILE], LABEL_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML video classification model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_pipeline:automl,video" + }, + "source": [ + "### Create a training pipeline\n", + "\n", + "You may ask, what do we use a pipeline for? You typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `pipeline_name`: A human readable name for the pipeline job.\n", + "- `model_name`: A human readable name for the model.\n", + "- `dataset`: The Vertex fully qualified dataset identifier.\n", + "- `schema`: The dataset labeling (annotation) training schema.\n", + "- `task`: A dictionary describing the requirements for the training job.\n", + "\n", + "The helper function calls the `Pipeline` client service'smethod `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "Let's look now deeper into the *minimal* requirements for constructing a `training_pipeline` specification:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The dataset labeling (annotation) training schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A human readable name for the model.\n", + "- `input_data_config`: The dataset specification.\n", + " - `dataset_id`: The Vertex dataset identifier only (non-fully qualified) -- this is the last part of the fully-qualified identifier.\n", + " - `fraction_split`: If specified, the percentages of the dataset to use for training, test and validation. Otherwise, the percentages are automatically selected by AutoML.\n", + " - Note for video, validation split is not supported -- only training and test." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_pipeline:automl,video" + }, + "outputs": [], + "source": [ + " def create_pipeline(pipeline_name, model_name, dataset, schema, task):\n", + "\n", + " dataset_id = dataset.split('/')[-1]\n", + "\n", + " input_config = {'dataset_id': dataset_id,\n", + " 'fraction_split': {\n", + " 'training_fraction': 0.8,\n", + " 'test_fraction': 0.2\n", + " }}\n", + "\n", + " training_pipeline = {\n", + " \"display_name\": pipeline_name,\n", + " \"training_task_definition\": schema,\n", + " \"training_task_inputs\": task,\n", + " \"input_data_config\": input_config,\n", + " \"model_to_upload\": {\"display_name\": model_name},\n", + " }\n", + "\n", + " try:\n", + " pipeline = clients['pipeline'].create_training_pipeline(parent=PARENT, training_pipeline=training_pipeline)\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "task_requirements:automl,vcn" + }, + "source": [ + "### Construct the task requirements\n", + "\n", + "Next, construct the task requirements. Unlike other parameters which take a Python (JSON-like) dictionary, the `task` field takes a Google protobuf Struct, which is very similar to a Python dictionary. Use the `json_format.ParseDict` method for the conversion.\n", + "\n", + "For video classification, there are no required minimal fields to specify.\n", + "\n", + "Finally, you create the pipeline by calling the helper function `create_pipeline`, which returns an instance of a training pipeline object." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_requirements:automl,vcn" + }, + "outputs": [], + "source": [ + "PIPE_NAME = \"hmdb,tst_pipe-\" + TIMESTAMP\n", + "MODEL_NAME = \"hmdb,tst_model-\" + TIMESTAMP\n", + "\n", + "task = json_format.ParseDict({}, Value())\n", + "\n", + "response = create_pipeline(PIPE_NAME, MODEL_NAME, dataset_id, TRAINING_SCHEMA, task)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split('/')[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients['pipeline'].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "model_information" + }, + "source": [ + "## Model information\n", + "\n", + "Now that your model is trained, you can get some information on your model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:automl" + }, + "source": [ + "## Evaluate the Model resource\n", + "\n", + "Now find out how good the model service believes your model is. As part of training, some portion of the dataset was set aside as the test (holdout) data, which is used by the pipeline service to evaluate the model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "list_model_evaluations:automl,vcn" + }, + "source": [ + "### List evaluations for all slices\n", + "\n", + "Use this helper function `list_model_evaluations`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified model identifier for the `Model` resource.\n", + "\n", + "This helper function uses the model client service's `list_model_evaluations` method, which takes the same parameter. The response object from the call is a list, where each element is an evaluation metric.\n", + "\n", + "For each evaluation (you probably only have one) we then print all the key names for each metric in the evaluation, and for a small set (`auPrc`) you will print the result." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "list_model_evaluations:automl,vcn" + }, + "outputs": [], + "source": [ + "def list_model_evaluations(name):\n", + " response = clients['model'].list_model_evaluations(parent=name)\n", + " for evaluation in response:\n", + " print(\"model_evaluation\")\n", + " print(\" name:\", evaluation.name)\n", + " print(\" metrics_schema_uri:\", evaluation.metrics_schema_uri)\n", + " metrics = json_format.MessageToDict(evaluation._pb.metrics)\n", + " for metric in metrics.keys():\n", + " print(metric)\n", + " print('auPrc', metrics['auPrc'])\n", + "\n", + "\n", + " return evaluation.name\n", + "\n", + "\n", + "last_evaluation = list_model_evaluations(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy:batch_prediction" + }, + "source": [ + "## Model deployment for batch prediction\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created for batch prediction. This differs from deploying a `Model` resource for on-demand prediction.\n", + "\n", + "For online prediction, you:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource.\n", + "\n", + "3. Make online prediction requests to the `Endpoint` resource.\n", + "\n", + "For batch-prediction, you:\n", + "\n", + "1. Create a batch prediction job.\n", + "\n", + "2. The job service will provision resources for the batch prediction request.\n", + "\n", + "3. The results of the batch prediction request are returned to the caller.\n", + "\n", + "4. The job service will unprovision the resoures for the batch prediction request." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a batch prediction request\n", + "\n", + "Now do a batch prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_items:batch_prediction" + }, + "source": [ + "### Get test item(s)\n", + "\n", + "Now do a batch prediction to your Vertex model. You will use arbitrary examples out of the dataset as a test items. Don't be concerned that the examples were likely used in training the model -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_items:automl,vcn,csv" + }, + "outputs": [], + "source": [ + "test_items = ! gsutil cat $IMPORT_FILE | head -n2\n", + "\n", + "if len(test_items[0]) == 5:\n", + " _, test_item_1, test_label_1, _, _ = str(test_items[0]).split(',')\n", + " _, test_item_2, test_label_2, _, _ = str(test_items[1]).split(',')\n", + "else:\n", + " test_item_1, test_label_1, _, _ = str(test_items[0]).split(',')\n", + " test_item_2, test_label_2, _, _ = str(test_items[1]).split(',')\n", + "\n", + "\n", + "print(test_item_1, test_label_1)\n", + "print(test_item_2, test_label_2)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:automl,video" + }, + "source": [ + "### Make a batch input file\n", + "\n", + "Now make a batch input file, which you store in your local Cloud Storage bucket. The batch input file can be either CSV or JSONL. You will use JSONL in this tutorial. For JSONL file, you make one dictionary entry per line for each video. The dictionary contains the key/value pairs:\n", + "\n", + "- `content`: The Cloud Storage path to the video.\n", + "- `mimeType`: The content type. In our example, it is an `avi` file.\n", + "- `timeSegmentStart`: The start timestamp in the video to do prediction on. *Note*, the timestamp must be specified as a string and followed by s (second), m (minute) or h (hour).\n", + "- `timeSegmentEnd`: The end timestamp in the video to do prediction on." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_file:automl,video" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "import tensorflow as tf\n", + "\n", + "gcs_input_uri = BUCKET_NAME + '/test.jsonl'\n", + "with tf.io.gfile.GFile(gcs_input_uri, 'w') as f:\n", + " data = { \"content\": test_item_1, \"mimeType\": \"video/avi\", \"timeSegmentStart\": \"0.0s\", 'timeSegmentEnd': '5.0s' }\n", + " f.write(json.dumps(data) + '\\n')\n", + " data = { \"content\": test_item_2, \"mimeType\": \"video/avi\", \"timeSegmentStart\": \"0.0s\", 'timeSegmentEnd': '5.0s' }\n", + " f.write(json.dumps(data) + '\\n')\n", + "\n", + "print(gcs_input_uri)\n", + "! gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your batch prediction requests:\n", + "\n", + "- Single Instance: The batch prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The batch prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and batch prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The batch prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_request:automl,vcn" + }, + "source": [ + "### Make batch prediction request\n", + "\n", + "Now that your batch of two test items is ready, let's do the batch request. Use this helper function `create_batch_prediction_job`, with the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the prediction job.\n", + "- `model_name`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `gcs_source_uri`: The Cloud Storage path to the input file -- which you created above.\n", + "- `gcs_destination_output_uri_prefix`: The Cloud Storage path that the service will write the predictions to.\n", + "- `parameters`: Additional filtering parameters for serving prediction results.\n", + "\n", + "The helper function calls the job client service's `create_batch_prediction_job` metho, with the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for Dataset, Model and Pipeline resources.\n", + "- `batch_prediction_job`: The specification for the batch prediction job.\n", + "\n", + "Let's now dive into the specification for the `batch_prediction_job`:\n", + "\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `dedicated_resources`: The compute resources to provision for the batch prediction job.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `starting_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "- `model_parameters`: Additional filtering parameters for serving prediction results.\n", + " - `confidenceThreshold`: The minimum confidence threshold on doing a prediction.\n", + " - `maxPredictions`: The maximum number of predictions to return per classification, sorted by confidence.\n", + " - `oneSecIntervalClassification`: If `True`, predictions are made on one second intervals.\n", + " - `shotClassification`: If `True`, predictions are made on each camera shot boundary.\n", + " - `segmentClassification`: If `True`, predictions are made on each time segment; otherwise prediction is made for the entire time segment.\n", + "- `input_config`: The input source and format type for the instances to predict.\n", + " - `instances_format`: The format of the batch prediction request file: `csv` or `jsonl`.\n", + " - `gcs_source`: A list of one or more Cloud Storage paths to your batch prediction requests.\n", + "- `output_config`: The output destination and format for the predictions.\n", + " - `prediction_format`: The format of the batch prediction response file: `jsonl` only.\n", + " - `gcs_destination`: The output destination for the predictions.\n", + "\n", + "You might ask, how does confidence_threshold affect the model accuracy? The threshold won't change the accuracy. What it changes is recall and precision.\n", + "\n", + " - Precision: The higher the precision the more likely what is predicted is the correct prediction, but return fewer predictions. Increasing the confidence threshold increases precision.\n", + " - Recall: The higher the recall the more likely a correct prediction is returned in the result, but return more prediction with incorrect prediction. Decreasing the confidence threshold increases recall.\n", + "\n", + "In this example, you will predict for precision. You set the confidence threshold to 0.5 and the maximum number of predictions for an action to two. Since, all the confidence values across the classes must add up to one, there are only two possible outcomes:\n", + "\n", + " 1. There is a tie, both 0.5, and returns two predictions.\n", + " 2. One value is above 0.5 and the rest are below 0.5, and returns one prediction.\n", + "\n", + "This call is an asychronous operation. You will print from the response object a few select fields, including:\n", + "\n", + "- `name`: The Vertex fully qualified identifier assigned to the batch prediction job.\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the Model resource.\n", + "- `generate_explanations`: Whether True/False explanations were provided with the predictions (explainability).\n", + "- `state`: The state of the prediction job (pending, running, etc).\n", + "\n", + "Since this call will take a few moments to execute, you will likely get `JobState.JOB_STATE_PENDING` for `state`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_request:automl,vcn" + }, + "outputs": [], + "source": [ + "BATCH_MODEL = \"hmdb,tst_batch-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_batch_prediction_job(display_name, model_name, gcs_source_uri, gcs_destination_output_uri_prefix, parameters=None):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " batch_prediction_job = {\n", + " \"display_name\": display_name,\n", + " # Format: 'projects/{project}/locations/{location}/models/{model_id}'\n", + " \"model\": model_name,\n", + " \"model_parameters\": json_format.ParseDict(parameters, Value()),\n", + " \"input_config\": {\n", + " \"instances_format\": IN_FORMAT,\n", + " \"gcs_source\": {\"uris\": [gcs_source_uri]},\n", + " },\n", + " \"output_config\": {\n", + " \"predictions_format\": OUT_FORMAT,\n", + " \"gcs_destination\": {\"output_uri_prefix\": gcs_destination_output_uri_prefix},\n", + " },\n", + " \"dedicated_resources\": {\n", + " \"machine_spec\": machine_spec,\n", + " \"starting_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES\n", + " }\n", + "\n", + " }\n", + " response = clients['job'].create_batch_prediction_job(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + " )\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try:\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", response.labels)\n", + " return response\n", + "\n", + "\n", + "IN_FORMAT = 'jsonl'\n", + "OUT_FORMAT = 'jsonl' # [jsonl]\n", + "\n", + "response = create_batch_prediction_job(BATCH_MODEL, model_to_deploy_id, gcs_input_uri, BUCKET_NAME, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batch_job_id:response" + }, + "source": [ + "Now get the unique identifier for the batch prediction job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batch_job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the batch job\n", + "batch_job_id = response.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split('/')[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_batch_prediction_job" + }, + "source": [ + "### Get information on a batch prediction job\n", + "\n", + "Use this helper function `get_batch_prediction_job`, with the following paramter:\n", + "\n", + "- `job_name`: The Vertex fully qualified identifier for the batch prediction job.\n", + "\n", + "The helper function calls the job client service's `get_batch_prediction_job` method, with the following paramter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the batch prediction job. In this tutorial, you will pass it the Vertex fully qualified identifier for your batch prediction job -- `batch_job_id`\n", + "\n", + "The helper function will return the Cloud Storage path to where the predictions are stored -- `gcs_destination`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_batch_prediction_job" + }, + "outputs": [], + "source": [ + "def get_batch_prediction_job(job_name, silent=False):\n", + " response = clients['job'].get_batch_prediction_job(name=job_name)\n", + " if silent:\n", + " return response.output_config.gcs_destination.output_uri_prefix, response.state\n", + "\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try: # not all data types support explanations\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" error:\", response.error)\n", + " gcs_destination = response.output_config.gcs_destination\n", + " print(\" gcs_destination\")\n", + " print(\" output_uri_prefix:\", gcs_destination.output_uri_prefix)\n", + " return gcs_destination.output_uri_prefix, response.state\n", + "\n", + "\n", + "predictions, state = get_batch_prediction_job(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_the_predictions:automl,vcn" + }, + "source": [ + "### Get the predictions\n", + "\n", + "When the batch prediction is done processing, the job state will be `JOB_STATE_SUCCEEDED`.\n", + "\n", + "Finally you view the predictions stored at the Cloud Storage path you set as output. The predictions will be in a JSONL format, which you indicated at the time you made the batch prediction job, under a subfolder starting with the name `prediction`, and under that folder will be a file called `predictions*.jsonl`.\n", + "\n", + "Now display (cat) the contents. You will see multiple JSON objects, one for each prediction.\n", + "\n", + "For each prediction:\n", + "\n", + "- `content`: The video that was input for the prediction request.\n", + "- `displayName`: The prediction action.\n", + "- `confidence`: The confidence in the prediction between 0 and 1.\n", + "- `timeSegmentStart/timeSegmentEnd`: The time offset of the start and end of the predicted action." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_the_predictions:automl,video" + }, + "outputs": [], + "source": [ + "def get_latest_predictions(gcs_out_dir):\n", + " ''' Get the latest prediction subfolder using the timestamp in the subfolder name'''\n", + " folders = !gsutil ls $gcs_out_dir\n", + " latest = \"\"\n", + " for folder in folders:\n", + " subfolder = folder.split('/')[-2]\n", + " if subfolder.startswith('prediction-'):\n", + " if subfolder > latest:\n", + " latest = folder[:-1]\n", + " return latest\n", + "\n", + "\n", + "while True:\n", + " predictions, state = get_batch_prediction_job(batch_job_id, True)\n", + " if state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", state)\n", + " if state == aip.JobState.JOB_STATE_FAILED:\n", + " raise Exception(\"Batch Job Failed\")\n", + " else:\n", + " folder = get_latest_predictions(predictions)\n", + " ! gsutil ls $folder/prediction*.jsonl\n", + "\n", + " ! gsutil cat $folder/prediction*.jsonl\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and 'dataset_id' in globals():\n", + " clients['dataset'].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and 'pipeline_id' in globals():\n", + " clients['pipeline'].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and 'model_to_deploy_id' in globals():\n", + " clients['model'].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and 'endpoint_id' in globals():\n", + " clients['endpoint'].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and 'batch_job_id' in globals():\n", + " clients['job'].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and 'job_id' in globals():\n", + " clients['job'].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and 'hpt_job_id' in globals():\n", + " clients['job'].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and 'BUCKET_NAME' in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_automl_video_classification_batch.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/automl/showcase_automl_video_object_tracking_batch.ipynb b/notebooks/community/gapic/automl/showcase_automl_video_object_tracking_batch.ipynb new file mode 100644 index 000000000..9509dbe45 --- /dev/null +++ b/notebooks/community/gapic/automl/showcase_automl_video_object_tracking_batch.ipynb @@ -0,0 +1,1772 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: AutoML video object tracking model for batch prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to create video object tracking models and do batch prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:traffic,vot" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Traffic](https://todo) from [??](todo). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,batch_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML video object tracking model from a Python script, and then do a batch prediction using the Vertex client library. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Make a batch prediction.\n", + "\n", + "There is one key difference between using batch prediction and using online prediction:\n", + "\n", + "* Prediction Service: Does an on-demand prediction for the entire set of instances (i.e., one or more data items) and returns the results in real-time.\n", + "\n", + "* Batch Prediction Service: Does a queued (batch) prediction for the entire set of instances in the background and stores the results in a Cloud Storage bucket when ready." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = '--user'\n", + "else:\n", + " USER_FLAG = ''\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" #@param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = 'us-central1' #@param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" #@param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Set constants unique to AutoML datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the `Dataset` resource service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the `Dataset` resource service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:vot" + }, + "outputs": [], + "source": [ + "# Video Dataset type\n", + "DATA_SCHEMA = 'gs://google-cloud-aiplatform/schema/dataset/metadata/video_1.0.0.yaml'\n", + "# Video Labeling type\n", + "LABEL_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/ioformat/video_object_tracking_io_format_1.0.0.yaml\"\n", + "# Video Training task\n", + "TRAINING_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_video_object_tracking_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:prediction" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for prediction.\n", + "\n", + "Set the variable `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")))\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:automl" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "For AutoML batch prediction, the container image for the serving binary is pre-determined by the Vertex prediction service. More specifically, the service will pick the appropriate container for the model depending on the hardware accelerator you selected." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for prediction.\n", + "\n", + "- Set the variable `DEPLOY_COMPUTE` to configure the compute resources for the VM you will use for prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = 'n1-standard'\n", + "\n", + "VCPU = '4'\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + '-' + VCPU\n", + "print('Deploy machine type', DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML video object tracking model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:automl,batch_prediction" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training.\n", + "- Job Service for batch prediction and custom training." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:automl,batch_prediction" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(\n", + " client_options=client_options\n", + " )\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(\n", + " client_options=client_options\n", + " )\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(\n", + " client_options=client_options\n", + " )\n", + " return client\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(\n", + " client_options=client_options\n", + " )\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients['dataset'] = create_dataset_client()\n", + "clients['model'] = create_model_client()\n", + "clients['pipeline'] = create_pipeline_client()\n", + "clients['job'] = create_job_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "def create_dataset(name, schema, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " dataset = aip.Dataset(display_name=name, metadata_schema_uri=schema, labels=labels)\n", + "\n", + " operation = clients['dataset'].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"traffic-\" + TIMESTAMP, DATA_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split('/')[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:video,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for video has some requirements for your data.\n", + "\n", + "- Videos must be stored in a Cloud Storage bucket.\n", + "- Each video file must be in a video format (MPG, AVI, ...).\n", + "- There must be an index file stored in your Cloud Storage bucket that contains the path and label for each video.\n", + "- The index file must be either CSV or JSONL." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:vot,u_dataset,csv" + }, + "source": [ + "#### CSV\n", + "\n", + "For video object tracking, the CSV index file has a few requirements:\n", + "\n", + "- No heading.\n", + "- First column is the Cloud Storage path to the video.\n", + "- Second column is the label.\n", + "- Third column is **not used**\n", + "- Fourth column is **not used**\n", + "- Fifth/Sixth columns are the upper left corner of bounding box. Coordinates are normalized, between 0 and 1.\n", + "- Seventh/Eigth/Ninth columns are not used and should be 0.\n", + "- Tenth/Eleventh columns are the lower right corner of the bounding box." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:traffic,csv,vot" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = 'gs://automl-video-demo-data/traffic_videos/traffic_videos_labels.csv'" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:csv" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Traffic dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:csv" + }, + "outputs": [], + "source": [ + "if 'IMPORT_FILES' in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_data" + }, + "source": [ + "### Import data\n", + "\n", + "Now, import the data into your Vertex Dataset resource. Use this helper function `import_data` to import the data. The function does the following:\n", + "\n", + "- Uses the `Dataset` client.\n", + "- Calls the client method `import_data`, with the following parameters:\n", + " - `name`: The human readable name you give to the `Dataset` resource (e.g., traffic).\n", + " - `import_configs`: The import configuration.\n", + "\n", + "- `import_configs`: A Python list containing a dictionary, with the key/value entries:\n", + " - `gcs_sources`: A list of URIs to the paths of the one or more index files.\n", + " - `import_schema_uri`: The schema identifying the labeling type.\n", + "\n", + "The `import_data()` method returns a long running `operation` object. This will take a few minutes to complete. If you are in a live tutorial, this would be a good time to ask questions, or take a personal break." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_data" + }, + "outputs": [], + "source": [ + "def import_data(dataset, gcs_sources, schema):\n", + " config = [{\n", + " 'gcs_source': {'uris': gcs_sources},\n", + " 'import_schema_uri': schema\n", + " }]\n", + " print(\"dataset:\", dataset_id)\n", + " start_time = time.time()\n", + " try:\n", + " operation = clients['dataset'].import_data(name=dataset_id, import_configs=config)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + "\n", + " result = operation.result()\n", + " print(\"result:\", result)\n", + " print(\"time:\", int(time.time() - start_time), \"secs\")\n", + " print(\"error:\", operation.exception())\n", + " print(\"meta :\", operation.metadata)\n", + " print(\"after: running:\", operation.running(), \"done:\", operation.done(), \"cancelled:\", operation.cancelled())\n", + "\n", + " return operation\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "import_data(dataset_id, [IMPORT_FILE], LABEL_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML video object tracking model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_pipeline:automl,video" + }, + "source": [ + "### Create a training pipeline\n", + "\n", + "You may ask, what do we use a pipeline for? You typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `pipeline_name`: A human readable name for the pipeline job.\n", + "- `model_name`: A human readable name for the model.\n", + "- `dataset`: The Vertex fully qualified dataset identifier.\n", + "- `schema`: The dataset labeling (annotation) training schema.\n", + "- `task`: A dictionary describing the requirements for the training job.\n", + "\n", + "The helper function calls the `Pipeline` client service'smethod `create_pipeline`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "Let's look now deeper into the *minimal* requirements for constructing a `training_pipeline` specification:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The dataset labeling (annotation) training schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A human readable name for the model.\n", + "- `input_data_config`: The dataset specification.\n", + " - `dataset_id`: The Vertex dataset identifier only (non-fully qualified) -- this is the last part of the fully-qualified identifier.\n", + " - `fraction_split`: If specified, the percentages of the dataset to use for training, test and validation. Otherwise, the percentages are automatically selected by AutoML.\n", + " - Note for video, validation split is not supported -- only training and test." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_pipeline:automl,video" + }, + "outputs": [], + "source": [ + " def create_pipeline(pipeline_name, model_name, dataset, schema, task):\n", + "\n", + " dataset_id = dataset.split('/')[-1]\n", + "\n", + " input_config = {'dataset_id': dataset_id,\n", + " 'fraction_split': {\n", + " 'training_fraction': 0.8,\n", + " 'test_fraction': 0.2\n", + " }}\n", + "\n", + " training_pipeline = {\n", + " \"display_name\": pipeline_name,\n", + " \"training_task_definition\": schema,\n", + " \"training_task_inputs\": task,\n", + " \"input_data_config\": input_config,\n", + " \"model_to_upload\": {\"display_name\": model_name},\n", + " }\n", + "\n", + " try:\n", + " pipeline = clients['pipeline'].create_training_pipeline(parent=PARENT, training_pipeline=training_pipeline)\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "task_requirements:automl,vot" + }, + "source": [ + "### Construct the task requirements\n", + "\n", + "Next, construct the task requirements. Unlike other parameters which take a Python (JSON-like) dictionary, the `task` field takes a Google protobuf Struct, which is very similar to a Python dictionary. Use the `json_format.ParseDict` method for the conversion.\n", + "\n", + "The minimal fields you need to specify are:\n", + "\n", + "- `model_type`: The type of deployed model, ex. CLOUD for deploying to Google Cloud.\n", + "\n", + "Finally, create the pipeline by calling the helper function `create_pipeline`, which returns an instance of a training pipeline object." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "task_requirements:automl,vot" + }, + "outputs": [], + "source": [ + "PIPE_NAME = \"traffic_pipe-\" + TIMESTAMP\n", + "MODEL_NAME = \"traffic_model-\" + TIMESTAMP\n", + "\n", + "task = json_format.ParseDict({'model_type': \"CLOUD\",\n", + " }, Value())\n", + "\n", + "response = create_pipeline(PIPE_NAME, MODEL_NAME, dataset_id, TRAINING_SCHEMA, task)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split('/')[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients['pipeline'].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 240 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "model_information" + }, + "source": [ + "## Model information\n", + "\n", + "Now that your model is trained, you can get some information on your model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:automl" + }, + "source": [ + "## Evaluate the Model resource\n", + "\n", + "Now find out how good the model service believes your model is. As part of training, some portion of the dataset was set aside as the test (holdout) data, which is used by the pipeline service to evaluate the model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "list_model_evaluations:automl,vot" + }, + "source": [ + "### List evaluations for all slices\n", + "\n", + "Use this helper function `list_model_evaluations`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified model identifier for the `Model` resource.\n", + "\n", + "This helper function uses the model client service's `list_model_evaluations` method, which takes the same parameter. The response object from the call is a list, where each element is an evaluation metric.\n", + "\n", + "For each evaluation -- you probably only have one, you then print all the key names for each metric in the evaluation, and for a small set (`boundingBoxMetrics`) you will print the result." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "list_model_evaluations:automl,vot" + }, + "outputs": [], + "source": [ + "def list_model_evaluations(name):\n", + " response = clients['model'].list_model_evaluations(parent=name)\n", + " for evaluation in response:\n", + " print(\"model_evaluation\")\n", + " print(\" name:\", evaluation.name)\n", + " print(\" metrics_schema_uri:\", evaluation.metrics_schema_uri)\n", + " metrics = json_format.MessageToDict(evaluation._pb.metrics)\n", + " for metric in metrics.keys():\n", + " print(metric)\n", + " print('boundingBoxMetrics', metrics['boundingBoxMetrics'])\n", + "\n", + "\n", + " return evaluation.name\n", + "\n", + "\n", + "last_evaluation = list_model_evaluations(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy:batch_prediction" + }, + "source": [ + "## Model deployment for batch prediction\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created for batch prediction. This differs from deploying a `Model` resource for on-demand prediction.\n", + "\n", + "For online prediction, you:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource.\n", + "\n", + "3. Make online prediction requests to the `Endpoint` resource.\n", + "\n", + "For batch-prediction, you:\n", + "\n", + "1. Create a batch prediction job.\n", + "\n", + "2. The job service will provision resources for the batch prediction request.\n", + "\n", + "3. The results of the batch prediction request are returned to the caller.\n", + "\n", + "4. The job service will unprovision the resoures for the batch prediction request." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a batch prediction request\n", + "\n", + "Now do a batch prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_items:batch_prediction" + }, + "source": [ + "### Get test item(s)\n", + "\n", + "Now do a batch prediction to your Vertex model. You will use arbitrary examples out of the dataset as a test items. Don't be concerned that the examples were likely used in training the model -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_items:automl,vot,csv" + }, + "outputs": [], + "source": [ + "test_items = ! gsutil cat $IMPORT_FILE | head -n2\n", + "\n", + "cols_1 = test_items[0].split(',')\n", + "cols_2 = test_items[1].split(',')\n", + "if len(cols_1) > 12:\n", + " test_item_1 = str(cols_1[1])\n", + " test_item_2 = str(cols_2[1])\n", + " test_label_1 = str(cols_1[2])\n", + " test_label_2 = str(cols_2[2])\n", + "else:\n", + " test_item_1 = str(cols_1[0])\n", + " test_item_2 = str(cols_2[0])\n", + " test_label_1 = str(cols_1[1])\n", + " test_label_2 = str(cols_2[1])\n", + "\n", + "\n", + "print(test_item_1, test_label_1)\n", + "print(test_item_2, test_label_2)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:automl,video" + }, + "source": [ + "### Make a batch input file\n", + "\n", + "Now make a batch input file, which you store in your local Cloud Storage bucket. The batch input file can be either CSV or JSONL. You will use JSONL in this tutorial. For JSONL file, you make one dictionary entry per line for each video. The dictionary contains the key/value pairs:\n", + "\n", + "- `content`: The Cloud Storage path to the video.\n", + "- `mimeType`: The content type. In our example, it is an `avi` file.\n", + "- `timeSegmentStart`: The start timestamp in the video to do prediction on. *Note*, the timestamp must be specified as a string and followed by s (second), m (minute) or h (hour).\n", + "- `timeSegmentEnd`: The end timestamp in the video to do prediction on." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_file:automl,video" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "import tensorflow as tf\n", + "\n", + "gcs_input_uri = BUCKET_NAME + '/test.jsonl'\n", + "with tf.io.gfile.GFile(gcs_input_uri, 'w') as f:\n", + " data = { \"content\": test_item_1, \"mimeType\": \"video/avi\", \"timeSegmentStart\": \"0.0s\", 'timeSegmentEnd': '5.0s' }\n", + " f.write(json.dumps(data) + '\\n')\n", + " data = { \"content\": test_item_2, \"mimeType\": \"video/avi\", \"timeSegmentStart\": \"0.0s\", 'timeSegmentEnd': '5.0s' }\n", + " f.write(json.dumps(data) + '\\n')\n", + "\n", + "print(gcs_input_uri)\n", + "! gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your batch prediction requests:\n", + "\n", + "- Single Instance: The batch prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The batch prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and batch prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The batch prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_request:automl,vot" + }, + "source": [ + "### Make batch prediction request\n", + "\n", + "Now that your batch of two test items is ready, let's do the batch request. Use this helper function `create_batch_prediction_job`, with the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the prediction job.\n", + "- `model_name`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `gcs_source_uri`: The Cloud Storage path to the input file -- which you created above.\n", + "- `gcs_destination_output_uri_prefix`: The Cloud Storage path that the service will write the predictions to.\n", + "- `parameters`: Additional filtering parameters for serving prediction results.\n", + "\n", + "The helper function calls the job client service's `create_batch_prediction_job` metho, with the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for Dataset, Model and Pipeline resources.\n", + "- `batch_prediction_job`: The specification for the batch prediction job.\n", + "\n", + "Let's now dive into the specification for the `batch_prediction_job`:\n", + "\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `dedicated_resources`: The compute resources to provision for the batch prediction job.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `starting_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "- `model_parameters`: Additional filtering parameters for serving prediction results.\n", + " - `confidenceThreshold`: The minimum confidence threshold on doing a prediction.\n", + " - `maxPredictions`: The maximum number of predictions to return per object, sorted by confidence.\n", + " - `minBoundingSize`: Minimum size of the shortest edge of the bounding box to return a classification.\n", + "- `input_config`: The input source and format type for the instances to predict.\n", + " - `instances_format`: The format of the batch prediction request file: `csv` or `jsonl`.\n", + " - `gcs_source`: A list of one or more Cloud Storage paths to your batch prediction requests.\n", + "- `output_config`: The output destination and format for the predictions.\n", + " - `prediction_format`: The format of the batch prediction response file: `jsonl` only supported.\n", + " - `gcs_destination`: The output destination for the predictions.\n", + "\n", + "You might ask, how does confidence_threshold affect the model accuracy? The threshold won't change the accuracy. What it changes is recall and precision.\n", + "\n", + " - Precision: The higher the precision the more likely what is predicted is the correct prediction, but return fewer predictions. Increasing the confidence threshold increases precision.\n", + " - Recall: The higher the recall the more likely a correct prediction is returned in the result, but return more prediction with incorrect prediction. Decreasing the confidence threshold increases recall.\n", + "\n", + "In this example, you will predict for precision. You set the confidence threshold to 0.5 and the maximum number of predictions for an action to two. Since, all the confidence values across the classes must add up to one, there are only two possible outcomes:\n", + "\n", + " 1. There is a tie, both 0.5, and returns two predictions.\n", + " 2. One value is above 0.5 and the rest are below 0.5, and returns one prediction.\n", + "\n", + "This call is an asychronous operation. You will print from the response object a few select fields, including:\n", + "\n", + "- `name`: The Vertex fully qualified identifier assigned to the batch prediction job.\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the Model resource.\n", + "- `generate_explanations`: Whether True/False explanations were provided with the predictions (explainability).\n", + "- `state`: The state of the prediction job (pending, running, etc).\n", + "\n", + "Since this call will take a few moments to execute, you will likely get `JobState.JOB_STATE_PENDING` for `state`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_request:automl,vot" + }, + "outputs": [], + "source": [ + "BATCH_MODEL = \"traffic_batch-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_batch_prediction_job(display_name, model_name, gcs_source_uri, gcs_destination_output_uri_prefix, parameters=None):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " batch_prediction_job = {\n", + " \"display_name\": display_name,\n", + " # Format: 'projects/{project}/locations/{location}/models/{model_id}'\n", + " \"model\": model_name,\n", + " \"model_parameters\": json_format.ParseDict(parameters, Value()),\n", + " \"input_config\": {\n", + " \"instances_format\": IN_FORMAT,\n", + " \"gcs_source\": {\"uris\": [gcs_source_uri]},\n", + " },\n", + " \"output_config\": {\n", + " \"predictions_format\": OUT_FORMAT,\n", + " \"gcs_destination\": {\"output_uri_prefix\": gcs_destination_output_uri_prefix},\n", + " },\n", + " \"dedicated_resources\": {\n", + " \"machine_spec\": machine_spec,\n", + " \"starting_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES\n", + " }\n", + "\n", + " }\n", + " response = clients['job'].create_batch_prediction_job(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + " )\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try:\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", response.labels)\n", + " return response\n", + "\n", + "\n", + "IN_FORMAT = 'jsonl'\n", + "OUT_FORMAT = 'jsonl' # [jsonl]\n", + "\n", + "response = create_batch_prediction_job(BATCH_MODEL, model_to_deploy_id, gcs_input_uri, BUCKET_NAME,\n", + " {'confidenceThreshold': 0.5, 'maxPredictions': 2})" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batch_job_id:response" + }, + "source": [ + "Now get the unique identifier for the batch prediction job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batch_job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the batch job\n", + "batch_job_id = response.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split('/')[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_batch_prediction_job" + }, + "source": [ + "### Get information on a batch prediction job\n", + "\n", + "Use this helper function `get_batch_prediction_job`, with the following paramter:\n", + "\n", + "- `job_name`: The Vertex fully qualified identifier for the batch prediction job.\n", + "\n", + "The helper function calls the job client service's `get_batch_prediction_job` method, with the following paramter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the batch prediction job. In this tutorial, you will pass it the Vertex fully qualified identifier for your batch prediction job -- `batch_job_id`\n", + "\n", + "The helper function will return the Cloud Storage path to where the predictions are stored -- `gcs_destination`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_batch_prediction_job" + }, + "outputs": [], + "source": [ + "def get_batch_prediction_job(job_name, silent=False):\n", + " response = clients['job'].get_batch_prediction_job(name=job_name)\n", + " if silent:\n", + " return response.output_config.gcs_destination.output_uri_prefix, response.state\n", + "\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try: # not all data types support explanations\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" error:\", response.error)\n", + " gcs_destination = response.output_config.gcs_destination\n", + " print(\" gcs_destination\")\n", + " print(\" output_uri_prefix:\", gcs_destination.output_uri_prefix)\n", + " return gcs_destination.output_uri_prefix, response.state\n", + "\n", + "\n", + "predictions, state = get_batch_prediction_job(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_the_predictions:automl,vot" + }, + "source": [ + "### Get Predictions\n", + "\n", + "When the batch prediction is done processing, the job state will be `JOB_STATE_SUCCEEDED`.\n", + "\n", + "Finally you view the predictions stored at the Cloud Storage path you set as output. The predictions will be in a CSV format, which you indicated at the time you made the batch prediction job, under a subfolder starting with the name `prediction`, and under that folder will be a file called `predictions*.csv`.\n", + "\n", + "Now display (cat) the contents. You will see multiple rows, one for each prediction.\n", + "\n", + "The first column is the video file you did the prediction on, and the last column is the prediction. In between are settings you specified as part of the prediction request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_the_predictions:automl,video" + }, + "outputs": [], + "source": [ + "def get_latest_predictions(gcs_out_dir):\n", + " ''' Get the latest prediction subfolder using the timestamp in the subfolder name'''\n", + " folders = !gsutil ls $gcs_out_dir\n", + " latest = \"\"\n", + " for folder in folders:\n", + " subfolder = folder.split('/')[-2]\n", + " if subfolder.startswith('prediction-'):\n", + " if subfolder > latest:\n", + " latest = folder[:-1]\n", + " return latest\n", + "\n", + "\n", + "while True:\n", + " predictions, state = get_batch_prediction_job(batch_job_id, True)\n", + " if state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", state)\n", + " if state == aip.JobState.JOB_STATE_FAILED:\n", + " raise Exception(\"Batch Job Failed\")\n", + " else:\n", + " folder = get_latest_predictions(predictions)\n", + " ! gsutil ls $folder/prediction*.jsonl\n", + "\n", + " ! gsutil cat $folder/prediction*.jsonl\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and 'dataset_id' in globals():\n", + " clients['dataset'].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and 'pipeline_id' in globals():\n", + " clients['pipeline'].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and 'model_to_deploy_id' in globals():\n", + " clients['model'].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and 'endpoint_id' in globals():\n", + " clients['endpoint'].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and 'batch_job_id' in globals():\n", + " clients['job'].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and 'job_id' in globals():\n", + " clients['job'].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and 'hpt_job_id' in globals():\n", + " clients['job'].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and 'BUCKET_NAME' in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_automl_video_object_tracking_batch.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_custom_image_classification_batch.ipynb b/notebooks/community/gapic/custom/showcase_custom_image_classification_batch.ipynb new file mode 100644 index 000000000..17bf55ae9 --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_custom_image_classification_batch.ipynb @@ -0,0 +1,2257 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Custom training image classification model for batch prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to train and deploy a custom image classification model for batch prediction." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:custom,cifar10,icn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [CIFAR10 dataset](https://www.tensorflow.org/datasets/catalog/cifar10) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use is built into TensorFlow. The trained model predicts which type of class an image is from ten classes: airplane, automobile, bird, cat, deer, dog, frog, horse, ship, truck." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,training,batch_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create a custom model, with a training pipeline, from a Python script in a Google prebuilt Docker container using the Vertex client library, and then do a batch prediction on the uploaded model. You can alternatively create custom models using `gcloud` command-line tool or online using Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex custom job for training a model.\n", + "- Train the TensorFlow model.\n", + "- Retrieve and load the model artifacts.\n", + "- View the model evaluation.\n", + "- Upload the model as a Vertex `Model` resource.\n", + "- Make a batch prediction." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for training and prediction.\n", + "\n", + "Set the variables `TRAIN_GPU/TRAIN_NGPU` and `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU.\n", + "\n", + "*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3 -- which is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n", + " TRAIN_GPU, TRAIN_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n", + " )\n", + "else:\n", + " TRAIN_GPU, TRAIN_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (None, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:training,prediction" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for training and prediction\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-3:latest`\n", + " - TensorFlow 2.4\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-4:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/training/xgboost-cpu.1-1`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/training/scikit-learn-cpu.0-23:latest`\n", + " - Pytorch\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-5:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-6:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-7:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for training](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers).\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-3:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-90:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-82:latest`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-23:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-22:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-20:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for prediction](https://cloud.google.com/vertex-ai/docs/predictions/pre-built-containers)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf2-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf2-cpu.{}\".format(TF)\n", + "else:\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "TRAIN_IMAGE = \"gcr.io/cloud-aiplatform/training/{}:latest\".format(TRAIN_VERSION)\n", + "DEPLOY_IMAGE = \"gcr.io/cloud-aiplatform/prediction/{}:latest\".format(DEPLOY_VERSION)\n", + "\n", + "print(\"Training:\", TRAIN_IMAGE, TRAIN_GPU, TRAIN_NGPU)\n", + "print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU, DEPLOY_NGPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:training,prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for training and prediction.\n", + "\n", + "- Set the variables `TRAIN_COMPUTE` and `DEPLOY_COMPUTE` to configure the compute resources for the VMs you will use for for training and prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: The following is not supported for training:*\n", + "\n", + " - `standard`: 2 vCPUs\n", + " - `highcpu`: 2, 4 and 8 vCPUs\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_TRAIN_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Train machine type\", TRAIN_COMPUTE)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own custom model and training for CIFAR10." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:custom" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Model Service for `Model` resources.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:custom" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"job\"] = create_job_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_model" + }, + "source": [ + "## Train a model\n", + "\n", + "There are two ways you can train a custom model using a container image:\n", + "\n", + "- **Use a Google Cloud prebuilt container**. If you use a prebuilt container, you will additionally specify a Python package to install into the container image. This Python package contains your code for training a custom model.\n", + "\n", + "- **Use your own custom container image**. If you use your own container, the container needs to contain your code for training a custom model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_specification:prebuilt_container" + }, + "source": [ + "## Prepare your custom job specification\n", + "\n", + "Now that your clients are ready, your first step is to create a Job Specification for your custom training job. The job specification will consist of the following:\n", + "\n", + "- `worker_pool_spec` : The specification of the type of machine(s) you will use for training and how many (single or distributed)\n", + "- `python_package_spec` : The specification of the Python package to be installed with the pre-built container." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "source": [ + "### Prepare your machine specification\n", + "\n", + "Now define the machine specification for your custom training job. This tells Vertex what type of machine instance to provision for the training.\n", + " - `machine_type`: The type of GCP instance to provision -- e.g., n1-standard-8.\n", + " - `accelerator_type`: The type, if any, of hardware accelerator. In this tutorial if you previously set the variable `TRAIN_GPU != None`, you are using a GPU; otherwise you will use a CPU.\n", + " - `accelerator_count`: The number of accelerators." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "outputs": [], + "source": [ + "if TRAIN_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": TRAIN_COMPUTE,\n", + " \"accelerator_type\": TRAIN_GPU,\n", + " \"accelerator_count\": TRAIN_NGPU,\n", + " }\n", + "else:\n", + " machine_spec = {\"machine_type\": TRAIN_COMPUTE, \"accelerator_count\": 0}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "source": [ + "### Prepare your disk specification\n", + "\n", + "(optional) Now define the disk specification for your custom training job. This tells Vertex what type and size of disk to provision in each machine instance for the training.\n", + "\n", + " - `boot_disk_type`: Either SSD or Standard. SSD is faster, and Standard is less expensive. Defaults to SSD.\n", + " - `boot_disk_size_gb`: Size of disk in GB." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "outputs": [], + "source": [ + "DISK_TYPE = \"pd-ssd\" # [ pd-ssd, pd-standard]\n", + "DISK_SIZE = 200 # GB\n", + "\n", + "disk_spec = {\"boot_disk_type\": DISK_TYPE, \"boot_disk_size_gb\": DISK_SIZE}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container" + }, + "source": [ + "### Define the worker pool specification\n", + "\n", + "Next, you define the worker pool specification for your custom training job. The worker pool specification will consist of the following:\n", + "\n", + "- `replica_count`: The number of instances to provision of this machine type.\n", + "- `machine_spec`: The hardware specification.\n", + "- `disk_spec` : (optional) The disk storage specification.\n", + "\n", + "- `python_package`: The Python training package to install on the VM instance(s) and which Python module to invoke, along with command line arguments for the Python module.\n", + "\n", + "Let's dive deeper now into the python package specification:\n", + "\n", + "-`executor_image_spec`: This is the docker image which is configured for your custom training job.\n", + "\n", + "-`package_uris`: This is a list of the locations (URIs) of your python training packages to install on the provisioned instance. The locations need to be in a Cloud Storage bucket. These can be either individual python files or a zip (archive) of an entire package. In the later case, the job service will unzip (unarchive) the contents into the docker image.\n", + "\n", + "-`python_module`: The Python module (script) to invoke for running the custom training job. In this example, you will be invoking `trainer.task.py` -- note that it was not neccessary to append the `.py` suffix.\n", + "\n", + "-`args`: The command line arguments to pass to the corresponding Pythom module. In this example, you will be setting:\n", + " - `\"--model-dir=\" + MODEL_DIR` : The Cloud Storage location where to store the model artifacts. There are two ways to tell the training script where to save the model artifacts:\n", + " - direct: You pass the Cloud Storage location as a command line argument to your training script (set variable `DIRECT = True`), or\n", + " - indirect: The service passes the Cloud Storage location as the environment variable `AIP_MODEL_DIR` to your training script (set variable `DIRECT = False`). In this case, you tell the service the model artifact location in the job specification.\n", + " - `\"--epochs=\" + EPOCHS`: The number of epochs for training.\n", + " - `\"--steps=\" + STEPS`: The number of steps (batches) per epoch.\n", + " - `\"--distribute=\" + TRAIN_STRATEGY\"` : The training distribution strategy to use for single or distributed training.\n", + " - `\"single\"`: single device.\n", + " - `\"mirror\"`: all GPU devices on a single compute instance.\n", + " - `\"multi\"`: all GPU devices on all compute instances." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_job_\" + TIMESTAMP\n", + "MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, JOB_NAME)\n", + "\n", + "if not TRAIN_NGPU or TRAIN_NGPU < 2:\n", + " TRAIN_STRATEGY = \"single\"\n", + "else:\n", + " TRAIN_STRATEGY = \"mirror\"\n", + "\n", + "EPOCHS = 20\n", + "STEPS = 100\n", + "\n", + "DIRECT = True\n", + "if DIRECT:\n", + " CMDARGS = [\n", + " \"--model-dir=\" + MODEL_DIR,\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "else:\n", + " CMDARGS = [\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "\n", + "worker_pool_spec = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": machine_spec,\n", + " \"disk_spec\": disk_spec,\n", + " \"python_package_spec\": {\n", + " \"executor_image_uri\": TRAIN_IMAGE,\n", + " \"package_uris\": [BUCKET_NAME + \"/trainer_cifar10.tar.gz\"],\n", + " \"python_module\": \"trainer.task\",\n", + " \"args\": CMDARGS,\n", + " },\n", + " }\n", + "]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "assemble_custom_job_specification" + }, + "source": [ + "### Assemble a job specification\n", + "\n", + "Now assemble the complete description for the custom job specification:\n", + "\n", + "- `display_name`: The human readable name you assign to this custom job.\n", + "- `job_spec`: The specification for the custom job.\n", + " - `worker_pool_specs`: The specification for the machine VM instances.\n", + " - `base_output_directory`: This tells the service the Cloud Storage location where to save the model artifacts (when variable `DIRECT = False`). The service will then pass the location to the training script as the environment variable `AIP_MODEL_DIR`, and the path will be of the form:\n", + "\n", + " /model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "assemble_custom_job_specification" + }, + "outputs": [], + "source": [ + "if DIRECT:\n", + " job_spec = {\"worker_pool_specs\": worker_pool_spec}\n", + "else:\n", + " job_spec = {\n", + " \"worker_pool_specs\": worker_pool_spec,\n", + " \"base_output_directory\": {\"output_uri_prefix\": MODEL_DIR},\n", + " }\n", + "\n", + "custom_job = {\"display_name\": JOB_NAME, \"job_spec\": job_spec}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "examine_training_package" + }, + "source": [ + "### Examine the training package\n", + "\n", + "#### Package layout\n", + "\n", + "Before you start the training, you will look at how a Python package is assembled for a custom training job. When unarchived, the package contains the following directory/file layout.\n", + "\n", + "- PKG-INFO\n", + "- README.md\n", + "- setup.cfg\n", + "- setup.py\n", + "- trainer\n", + " - \\_\\_init\\_\\_.py\n", + " - task.py\n", + "\n", + "The files `setup.cfg` and `setup.py` are the instructions for installing the package into the operating environment of the Docker image.\n", + "\n", + "The file `trainer/task.py` is the Python script for executing the custom training job. *Note*, when we referred to it in the worker pool specification, we replace the directory slash with a dot (`trainer.task`) and dropped the file suffix (`.py`).\n", + "\n", + "#### Package Assembly\n", + "\n", + "In the following cells, you will assemble the training package." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "examine_training_package" + }, + "outputs": [], + "source": [ + "# Make folder for Python training script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\ntag_build =\\n\\ntag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'tensorflow_datasets==1.3.0',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\nName: CIFAR10 image classification\\n\\nVersion: 0.0.0\\n\\nSummary: Demostration training script\\n\\nHome-page: www.google.com\\n\\nAuthor: Google\\n\\nAuthor-email: aferlitsch@google.com\\n\\nLicense: Public\\n\\nDescription: Demo\\n\\nPlatform: Vertex\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:cifar10" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the training script task.py. We won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n", + "- Loads CIFAR10 dataset from TF Datasets (tfds).\n", + "- Builds a model using TF.Keras model API.\n", + "- Compiles the model (`compile()`).\n", + "- Sets a training distribution strategy according to the argument `args.distribute`.\n", + "- Trains the model (`fit()`) with epochs and steps according to the arguments `args.epochs` and `args.steps`\n", + "- Saves the trained model (`save(args.model_dir)`) to the specified model directory." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:cifar10" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Single, Mirror and Multi-Machine Distributed Training for CIFAR-10\n", + "\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "import argparse\n", + "import os\n", + "import sys\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv(\"AIP_MODEL_DIR\"), type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.01, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=10, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=200, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "print('DEVICES', device_lib.list_local_devices())\n", + "\n", + "# Single Machine, single compute device\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "# Single Machine, multiple compute device\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "# Multiple Machine, multiple compute device\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "# Multi-worker configuration\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "# Preparing dataset\n", + "BUFFER_SIZE = 10000\n", + "BATCH_SIZE = 64\n", + "\n", + "\n", + "def make_datasets_unbatched():\n", + "\n", + " # Scaling CIFAR10 data from (0, 255] to (0., 1.]\n", + " def scale(image, label):\n", + " image = tf.cast(image, tf.float32)\n", + " image /= 255.0\n", + " return image, label\n", + "\n", + "\n", + " datasets, info = tfds.load(name='cifar10',\n", + " with_info=True,\n", + " as_supervised=True)\n", + " return datasets['train'].map(scale).cache().shuffle(BUFFER_SIZE).repeat()\n", + "\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_cnn_model():\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu', input_shape=(32, 32, 3)),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu'),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Flatten(),\n", + " tf.keras.layers.Dense(10, activation='softmax')\n", + " ])\n", + " model.compile(\n", + " loss=tf.keras.losses.sparse_categorical_crossentropy,\n", + " optimizer=tf.keras.optimizers.SGD(learning_rate=args.lr),\n", + " metrics=['accuracy'])\n", + " return model\n", + "\n", + "\n", + "# Train the model\n", + "NUM_WORKERS = strategy.num_replicas_in_sync\n", + "# Here the batch size scales up by number of workers since\n", + "# `tf.data.Dataset.batch` expects the global batch size.\n", + "GLOBAL_BATCH_SIZE = BATCH_SIZE * NUM_WORKERS\n", + "train_dataset = make_datasets_unbatched().batch(GLOBAL_BATCH_SIZE)\n", + "\n", + "with strategy.scope():\n", + " # Creation of dataset, and model building/compiling need to be within\n", + " # `strategy.scope()`.\n", + " model = build_and_compile_cnn_model()\n", + "\n", + "model.fit(x=train_dataset, epochs=args.epochs, steps_per_epoch=args.steps)\n", + "model.save(args.model_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tarball_training_script" + }, + "source": [ + "#### Store training script on your Cloud Storage bucket\n", + "\n", + "Next, you package the training folder into a compressed tar ball, and then store it in your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tarball_training_script" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_cifar10.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job" + }, + "source": [ + "### Train the model\n", + "\n", + "\n", + "Now start the training of your custom training job on Vertex. Use this helper function `create_custom_job`, which takes the following parameter:\n", + "\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "The helper function calls job client service's `create_custom_job` method, with the following parameters:\n", + "\n", + "-`parent`: The Vertex location path to `Dataset`, `Model` and `Endpoint` resources.\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "You will display a handful of the fields returned in `response` object, with the two that are of most interest are:\n", + "\n", + "`response.name`: The Vertex fully qualified identifier assigned to this custom training job. You save this identifier for using in subsequent steps.\n", + "\n", + "`response.state`: The current state of the custom training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job" + }, + "outputs": [], + "source": [ + "def create_custom_job(custom_job):\n", + " response = clients[\"job\"].create_custom_job(parent=PARENT, custom_job=custom_job)\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = create_custom_job(custom_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "job_id:response" + }, + "source": [ + "Now get the unique identifier for the custom job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom job\n", + "job_id = response.name\n", + "# The short numeric ID for the custom job\n", + "job_short_id = job_id.split(\"/\")[-1]\n", + "\n", + "print(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_custom_job" + }, + "source": [ + "### Get information on a custom job\n", + "\n", + "Next, use this helper function `get_custom_job`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "The helper function calls the job client service's`get_custom_job` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "If you recall, you got the Vertex fully qualified identifier for the custom job in the `response.name` field when you called the `create_custom_job` method, and saved the identifier in the variable `job_id`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_custom_job" + }, + "outputs": [], + "source": [ + "def get_custom_job(name, silent=False):\n", + " response = clients[\"job\"].get_custom_job(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = get_custom_job(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete:custom" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, we will need to know the location of the saved model, which the Python script saved in your local Cloud Storage bucket at `MODEL_DIR + '/saved_model.pb'`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete:custom" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_custom_job(job_id, True)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_path_to_deploy = None\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " if not DIRECT:\n", + " MODEL_DIR = MODEL_DIR + \"/model\"\n", + " model_path_to_deploy = MODEL_DIR\n", + " print(\"Training Time:\", response.update_time - response.create_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model_to_deploy:\", model_path_to_deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "load_saved_model" + }, + "source": [ + "## Load the saved model\n", + "\n", + "Your model is stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Now load it from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n", + "\n", + "To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "load_saved_model" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:image" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load the CIFAR10 test (holdout) data from `tf.keras.datasets`, using the method `load_data()`. This will return the dataset as a tuple of two elements. The first element is the training data and the second is the test data. Each element is also a tuple of two elements: the image data, and the corresponding labels.\n", + "\n", + "You don't need the training data, and hence why we loaded it as `(_, _)`.\n", + "\n", + "Before you can run the data through evaluation, you need to preprocess it:\n", + "\n", + "x_test:\n", + "1. Normalize (rescaling) the pixel data by dividing each pixel by 255. This will replace each single byte integer pixel with a 32-bit floating point number between 0 and 1.\n", + "\n", + "y_test:
\n", + "2. The labels are currently scalar (sparse). If you look back at the `compile()` step in the `trainer/task.py` script, you will find that it was compiled for sparse labels. So we don't need to do anything more." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "evaluate_custom_model:image,cifar10" + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "from tensorflow.keras.datasets import cifar10\n", + "\n", + "(_, _), (x_test, y_test) = cifar10.load_data()\n", + "x_test = (x_test / 255.0).astype(np.float32)\n", + "\n", + "print(x_test.shape, y_test.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "perform_evaluation_custom" + }, + "source": [ + "### Perform the model evaluation\n", + "\n", + "Now evaluate how well the model in the custom job did." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "perform_evaluation_custom" + }, + "outputs": [], + "source": [ + "model.evaluate(x_test, y_test)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "how_serving_function_works" + }, + "source": [ + "## Upload the model for serving\n", + "\n", + "Next, you will upload your TF.Keras model from the custom job to Vertex `Model` service, which will create a Vertex `Model` resource for your custom model. During upload, you need to define a serving function to convert data to the format your model expects. If you send encoded data to Vertex, your serving function ensures that the data is decoded on the model server before it is passed as input to your model.\n", + "\n", + "### How does the serving function work\n", + "\n", + "When you send a request to an online prediction server, the request is received by a HTTP server. The HTTP server extracts the prediction request from the HTTP request content body. The extracted prediction request is forwarded to the serving function. For Google pre-built prediction containers, the request content is passed to the serving function as a `tf.string`.\n", + "\n", + "The serving function consists of two parts:\n", + "\n", + "- `preprocessing function`:\n", + " - Converts the input (`tf.string`) to the input shape and data type of the underlying model (dynamic graph).\n", + " - Performs the same preprocessing of the data that was done during training the underlying model -- e.g., normalizing, scaling, etc.\n", + "- `post-processing function`:\n", + " - Converts the model output to format expected by the receiving application -- e.q., compresses the output.\n", + " - Packages the output for the the receiving application -- e.g., add headings, make JSON object, etc.\n", + "\n", + "Both the preprocessing and post-processing functions are converted to static graphs which are fused to the model. The output from the underlying model is passed to the post-processing function. The post-processing function passes the converted/packaged output back to the HTTP server. The HTTP server returns the output as the HTTP response content.\n", + "\n", + "One consideration you need to consider when building serving functions for TF.Keras models is that they run as static graphs. That means, you cannot use TF graph operations that require a dynamic graph. If you do, you will get an error during the compile of the serving function which will indicate that you are using an EagerTensor which is not supported." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_image" + }, + "source": [ + "### Serving function for image data\n", + "\n", + "To pass images to the prediction service, you encode the compressed (e.g., JPEG) image bytes into base 64 -- which makes the content safe from modification while transmitting binary data over the network. Since this deployed model expects input data as raw (uncompressed) bytes, you need to ensure that the base 64 encoded data gets converted back to raw bytes before it is passed as input to the deployed model.\n", + "\n", + "To resolve this, define a serving function (`serving_fn`) and attach it to the model as a preprocessing step. Add a `@tf.function` decorator so the serving function is fused to the underlying model (instead of upstream on a CPU).\n", + "\n", + "When you send a prediction or explanation request, the content of the request is base 64 decoded into a Tensorflow string (`tf.string`), which is passed to the serving function (`serving_fn`). The serving function preprocesses the `tf.string` into raw (uncompressed) numpy bytes (`preprocess_fn`) to match the input requirements of the model:\n", + "- `io.decode_jpeg`- Decompresses the JPG image which is returned as a Tensorflow tensor with three channels (RGB).\n", + "- `image.convert_image_dtype` - Changes integer pixel values to float 32.\n", + "- `image.resize` - Resizes the image to match the input shape for the model.\n", + "- `resized / 255.0` - Rescales (normalization) the pixel data between 0 and 1.\n", + "\n", + "At this point, the data can be passed to the model (`m_call`)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_image" + }, + "outputs": [], + "source": [ + "CONCRETE_INPUT = \"numpy_inputs\"\n", + "\n", + "\n", + "def _preprocess(bytes_input):\n", + " decoded = tf.io.decode_jpeg(bytes_input, channels=3)\n", + " decoded = tf.image.convert_image_dtype(decoded, tf.float32)\n", + " resized = tf.image.resize(decoded, size=(32, 32))\n", + " rescale = tf.cast(resized / 255.0, tf.float32)\n", + " return rescale\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def preprocess_fn(bytes_inputs):\n", + " decoded_images = tf.map_fn(\n", + " _preprocess, bytes_inputs, dtype=tf.float32, back_prop=False\n", + " )\n", + " return {\n", + " CONCRETE_INPUT: decoded_images\n", + " } # User needs to make sure the key matches model's input\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def serving_fn(bytes_inputs):\n", + " images = preprocess_fn(bytes_inputs)\n", + " prob = m_call(**images)\n", + " return prob\n", + "\n", + "\n", + "m_call = tf.function(model.call).get_concrete_function(\n", + " [tf.TensorSpec(shape=[None, 32, 32, 3], dtype=tf.float32, name=CONCRETE_INPUT)]\n", + ")\n", + "\n", + "tf.saved_model.save(\n", + " model, model_path_to_deploy, signatures={\"serving_default\": serving_fn}\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_signature:image" + }, + "source": [ + "## Get the serving function signature\n", + "\n", + "You can get the signatures of your model's input and output layers by reloading the model into memory, and querying it for the signatures corresponding to each layer.\n", + "\n", + "For your purpose, you need the signature of the serving function. Why? Well, when we send our data for prediction as a HTTP request packet, the image data is base64 encoded, and our TF.Keras model takes numpy input. Your serving function will do the conversion from base64 to a numpy array.\n", + "\n", + "When making a prediction request, you need to route the request to the serving function instead of the model, so you need to know the input layer name of the serving function -- which you will use later when you make a prediction request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_signature:image" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_path_to_deploy)\n", + "\n", + "serving_input = list(\n", + " loaded.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "print(\"Serving function input:\", serving_input)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "upload_the_model" + }, + "source": [ + "### Upload the model\n", + "\n", + "Use this helper function `upload_model` to upload your model, stored in SavedModel format, up to the `Model` service, which will instantiate a Vertex `Model` resource instance for your model. Once you've done that, you can use the `Model` resource instance in the same way as any other Vertex `Model` resource instance, such as deploying to an `Endpoint` resource for serving predictions.\n", + "\n", + "The helper function takes the following parameters:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` service.\n", + "- `image_uri`: The container image for the model deployment.\n", + "- `model_uri`: The Cloud Storage path to our SavedModel artificat. For this tutorial, this is the Cloud Storage location where the `trainer/task.py` saved the model artifacts, which we specified in the variable `MODEL_DIR`.\n", + "\n", + "The helper function calls the `Model` client service's method `upload_model`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for `Dataset`, `Model` and `Endpoint` resources.\n", + "- `model`: The specification for the Vertex `Model` resource instance.\n", + "\n", + "Let's now dive deeper into the Vertex model specification `model`. This is a dictionary object that consists of the following fields:\n", + "\n", + "- `display_name`: A human readable name for the `Model` resource.\n", + "- `metadata_schema_uri`: Since your model was built without an Vertex `Dataset` resource, you will leave this blank (`''`).\n", + "- `artificat_uri`: The Cloud Storage path where the model is stored in SavedModel format.\n", + "- `container_spec`: This is the specification for the Docker container that will be installed on the `Endpoint` resource, from which the `Model` resource will serve predictions. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + "\n", + "Uploading a model into a Vertex Model resource returns a long running operation, since it may take a few moments. You call response.result(), which is a synchronous call and will return when the Vertex Model resource is ready.\n", + "\n", + "The helper function returns the Vertex fully qualified identifier for the corresponding Vertex Model instance upload_model_response.model. You will save the identifier for subsequent steps in the variable model_to_deploy_id." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "upload_the_model" + }, + "outputs": [], + "source": [ + "IMAGE_URI = DEPLOY_IMAGE\n", + "\n", + "\n", + "def upload_model(display_name, image_uri, model_uri):\n", + " model = {\n", + " \"display_name\": display_name,\n", + " \"metadata_schema_uri\": \"\",\n", + " \"artifact_uri\": model_uri,\n", + " \"container_spec\": {\n", + " \"image_uri\": image_uri,\n", + " \"command\": [],\n", + " \"args\": [],\n", + " \"env\": [{\"name\": \"env_name\", \"value\": \"env_value\"}],\n", + " \"ports\": [{\"container_port\": 8080}],\n", + " \"predict_route\": \"\",\n", + " \"health_route\": \"\",\n", + " },\n", + " }\n", + " response = clients[\"model\"].upload_model(parent=PARENT, model=model)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " upload_model_response = response.result(timeout=180)\n", + " print(\"upload_model_response\")\n", + " print(\" model:\", upload_model_response.model)\n", + " return upload_model_response.model\n", + "\n", + "\n", + "model_to_deploy_id = upload_model(\n", + " \"cifar10-\" + TIMESTAMP, IMAGE_URI, model_path_to_deploy\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_model" + }, + "source": [ + "### Get `Model` resource information\n", + "\n", + "Now let's get the model information for just your model. Use this helper function `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource.\n", + "\n", + "This helper function calls the Vertex `Model` client service's method `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_model" + }, + "outputs": [], + "source": [ + "def get_model(name):\n", + " response = clients[\"model\"].get_model(name=name)\n", + " print(response)\n", + "\n", + "\n", + "get_model(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy:batch_prediction" + }, + "source": [ + "## Model deployment for batch prediction\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created for batch prediction. This differs from deploying a `Model` resource for on-demand prediction.\n", + "\n", + "For online prediction, you:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource.\n", + "\n", + "3. Make online prediction requests to the `Endpoint` resource.\n", + "\n", + "For batch-prediction, you:\n", + "\n", + "1. Create a batch prediction job.\n", + "\n", + "2. The job service will provision resources for the batch prediction request.\n", + "\n", + "3. The results of the batch prediction request are returned to the caller.\n", + "\n", + "4. The job service will unprovision the resoures for the batch prediction request." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a batch prediction request\n", + "\n", + "Now do a batch prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_items:test" + }, + "source": [ + "### Get test items\n", + "\n", + "You will use examples out of the test (holdout) portion of the dataset as a test items." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_items:test" + }, + "outputs": [], + "source": [ + "test_image_1 = x_test[0]\n", + "test_label_1 = y_test[0]\n", + "test_image_2 = x_test[1]\n", + "test_label_2 = y_test[1]\n", + "print(test_image_1.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "prepare_test_items:test,image" + }, + "source": [ + "### Prepare the request content\n", + "You are going to send the CIFAR10 images as compressed JPG image, instead of the raw uncompressed bytes:\n", + "\n", + "- `cv2.imwrite`: Use openCV to write the uncompressed image to disk as a compressed JPEG image.\n", + " - Denormalize the image data from \\[0,1) range back to [0,255).\n", + " - Convert the 32-bit floating point values to 8-bit unsigned integers." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "prepare_test_items:test,image" + }, + "outputs": [], + "source": [ + "import cv2\n", + "\n", + "cv2.imwrite(\"tmp1.jpg\", (test_image_1 * 255).astype(np.uint8))\n", + "cv2.imwrite(\"tmp2.jpg\", (test_image_2 * 255).astype(np.uint8))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "copy_test_items:test" + }, + "source": [ + "### Copy test item(s)\n", + "\n", + "For the batch prediction, you will copy the test items over to your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copy_test_items:test" + }, + "outputs": [], + "source": [ + "! gsutil cp tmp1.jpg $BUCKET_NAME/tmp1.jpg\n", + "! gsutil cp tmp2.jpg $BUCKET_NAME/tmp2.jpg\n", + "\n", + "test_item_1 = BUCKET_NAME + \"/tmp1.jpg\"\n", + "test_item_2 = BUCKET_NAME + \"/tmp2.jpg\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:custom,image" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "Now make a batch input file, which you will store in your local Cloud Storage bucket. The batch input file can only be in JSONL format. For JSONL file, you make one dictionary entry per line for each data item (instance). The dictionary contains the key/value pairs:\n", + "\n", + "- `input_name`: the name of the input layer of the underlying model.\n", + "- `'b64'`: A key that indicates the content is base64 encoded.\n", + "- `content`: The compressed JPG image bytes as a base64 encoded string.\n", + "\n", + "Each instance in the prediction request is a dictionary entry of the form:\n", + "\n", + " {serving_input: {'b64': content}}\n", + "\n", + "To pass the image data to the prediction service you encode the bytes into base64 -- which makes the content safe from modification when transmitting binary data over the network.\n", + "\n", + "- `tf.io.read_file`: Read the compressed JPG images into memory as raw bytes.\n", + "- `base64.b64encode`: Encode the raw bytes into a base64 encoded string." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_file:custom,image" + }, + "outputs": [], + "source": [ + "import base64\n", + "import json\n", + "\n", + "gcs_input_uri = BUCKET_NAME + \"/\" + \"test.jsonl\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " bytes = tf.io.read_file(test_item_1)\n", + " b64str = base64.b64encode(bytes.numpy()).decode(\"utf-8\")\n", + " data = {serving_input: {\"b64\": b64str}}\n", + " f.write(json.dumps(data) + \"\\n\")\n", + " bytes = tf.io.read_file(test_item_2)\n", + " b64str = base64.b64encode(bytes.numpy()).decode(\"utf-8\")\n", + " data = {serving_input: {\"b64\": b64str}}\n", + " f.write(json.dumps(data) + \"\\n\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your batch prediction requests:\n", + "\n", + "- Single Instance: The batch prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The batch prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and batch prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The batch prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_request:custom" + }, + "source": [ + "### Make batch prediction request\n", + "\n", + "Now that your batch of two test items is ready, let's do the batch request. Use this helper function `create_batch_prediction_job`, with the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the prediction job.\n", + "- `model_name`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `gcs_source_uri`: The Cloud Storage path to the input file -- which you created above.\n", + "- `gcs_destination_output_uri_prefix`: The Cloud Storage path that the service will write the predictions to.\n", + "- `parameters`: Additional filtering parameters for serving prediction results.\n", + "\n", + "The helper function calls the job client service's `create_batch_prediction_job` metho, with the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for Dataset, Model and Pipeline resources.\n", + "- `batch_prediction_job`: The specification for the batch prediction job.\n", + "\n", + "Let's now dive into the specification for the `batch_prediction_job`:\n", + "\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `dedicated_resources`: The compute resources to provision for the batch prediction job.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `starting_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "- `model_parameters`: Additional filtering parameters for serving prediction results. No Additional parameters are supported for custom models.\n", + "- `input_config`: The input source and format type for the instances to predict.\n", + " - `instances_format`: The format of the batch prediction request file: `csv` or `jsonl`.\n", + " - `gcs_source`: A list of one or more Cloud Storage paths to your batch prediction requests.\n", + "- `output_config`: The output destination and format for the predictions.\n", + " - `prediction_format`: The format of the batch prediction response file: `csv` or `jsonl`.\n", + " - `gcs_destination`: The output destination for the predictions.\n", + "\n", + "This call is an asychronous operation. You will print from the response object a few select fields, including:\n", + "\n", + "- `name`: The Vertex fully qualified identifier assigned to the batch prediction job.\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the Model resource.\n", + "- `generate_explanations`: Whether True/False explanations were provided with the predictions (explainability).\n", + "- `state`: The state of the prediction job (pending, running, etc).\n", + "\n", + "Since this call will take a few moments to execute, you will likely get `JobState.JOB_STATE_PENDING` for `state`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_request:custom" + }, + "outputs": [], + "source": [ + "BATCH_MODEL = \"cifar10_batch-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_batch_prediction_job(\n", + " display_name,\n", + " model_name,\n", + " gcs_source_uri,\n", + " gcs_destination_output_uri_prefix,\n", + " parameters=None,\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " batch_prediction_job = {\n", + " \"display_name\": display_name,\n", + " # Format: 'projects/{project}/locations/{location}/models/{model_id}'\n", + " \"model\": model_name,\n", + " \"model_parameters\": json_format.ParseDict(parameters, Value()),\n", + " \"input_config\": {\n", + " \"instances_format\": IN_FORMAT,\n", + " \"gcs_source\": {\"uris\": [gcs_source_uri]},\n", + " },\n", + " \"output_config\": {\n", + " \"predictions_format\": OUT_FORMAT,\n", + " \"gcs_destination\": {\"output_uri_prefix\": gcs_destination_output_uri_prefix},\n", + " },\n", + " \"dedicated_resources\": {\n", + " \"machine_spec\": machine_spec,\n", + " \"starting_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " },\n", + " }\n", + " response = clients[\"job\"].create_batch_prediction_job(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + " )\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try:\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", response.labels)\n", + " return response\n", + "\n", + "\n", + "IN_FORMAT = \"jsonl\"\n", + "OUT_FORMAT = \"jsonl\"\n", + "\n", + "response = create_batch_prediction_job(\n", + " BATCH_MODEL, model_to_deploy_id, gcs_input_uri, BUCKET_NAME\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batch_job_id:response" + }, + "source": [ + "Now get the unique identifier for the batch prediction job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batch_job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the batch job\n", + "batch_job_id = response.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split(\"/\")[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_batch_prediction_job" + }, + "source": [ + "### Get information on a batch prediction job\n", + "\n", + "Use this helper function `get_batch_prediction_job`, with the following paramter:\n", + "\n", + "- `job_name`: The Vertex fully qualified identifier for the batch prediction job.\n", + "\n", + "The helper function calls the job client service's `get_batch_prediction_job` method, with the following paramter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the batch prediction job. In this tutorial, you will pass it the Vertex fully qualified identifier for your batch prediction job -- `batch_job_id`\n", + "\n", + "The helper function will return the Cloud Storage path to where the predictions are stored -- `gcs_destination`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_batch_prediction_job" + }, + "outputs": [], + "source": [ + "def get_batch_prediction_job(job_name, silent=False):\n", + " response = clients[\"job\"].get_batch_prediction_job(name=job_name)\n", + " if silent:\n", + " return response.output_config.gcs_destination.output_uri_prefix, response.state\n", + "\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try: # not all data types support explanations\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" error:\", response.error)\n", + " gcs_destination = response.output_config.gcs_destination\n", + " print(\" gcs_destination\")\n", + " print(\" output_uri_prefix:\", gcs_destination.output_uri_prefix)\n", + " return gcs_destination.output_uri_prefix, response.state\n", + "\n", + "\n", + "predictions, state = get_batch_prediction_job(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_the_predictions:custom,icn" + }, + "source": [ + "### Get the predictions\n", + "\n", + "When the batch prediction is done processing, the job state will be `JOB_STATE_SUCCEEDED`.\n", + "\n", + "Finally you view the predictions stored at the Cloud Storage path you set as output. The predictions will be in a JSONL format, which you indicated at the time you made the batch prediction job, under a subfolder starting with the name `prediction`, and under that folder will be a file called `prediction.results-xxxxx-of-xxxxx`.\n", + "\n", + "Now display (cat) the contents. You will see multiple JSON objects, one for each prediction.\n", + "\n", + "The response contains a JSON object for each instance, in the form:\n", + "\n", + "{ 'instance': { 'bytes_input': { 'b64': .... } }, 'prediction': [ ... ] }\n", + "\n", + "- `instance`: The image data for which this prediction.\n", + "- `prediction`: The confidence level for each class." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_the_predictions:custom,image" + }, + "outputs": [], + "source": [ + "def get_latest_predictions(gcs_out_dir):\n", + " \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n", + " folders = !gsutil ls $gcs_out_dir\n", + " latest = \"\"\n", + " for folder in folders:\n", + " subfolder = folder.split(\"/\")[-2]\n", + " if subfolder.startswith(\"prediction-\"):\n", + " if subfolder > latest:\n", + " latest = folder[:-1]\n", + " return latest\n", + "\n", + "\n", + "while True:\n", + " predictions, state = get_batch_prediction_job(batch_job_id, True)\n", + " if state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", state)\n", + " if state == aip.JobState.JOB_STATE_FAILED:\n", + " raise Exception(\"Batch Job Failed\")\n", + " else:\n", + " folder = get_latest_predictions(predictions)\n", + " ! gsutil ls $folder/prediction.results*\n", + "\n", + " print(\"Results:\")\n", + " ! gsutil cat $folder/prediction.results*\n", + "\n", + " print(\"Errors:\")\n", + " ! gsutil cat $folder/prediction.errors*\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_custom_image_classification_batch.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_custom_image_classification_batch_explain.ipynb b/notebooks/community/gapic/custom/showcase_custom_image_classification_batch_explain.ipynb new file mode 100644 index 000000000..461270502 --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_custom_image_classification_batch_explain.ipynb @@ -0,0 +1,2465 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Custom training image classification model for batch prediction with explanation\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom,xai" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to train and deploy a custom image classification model for batch prediction with explanation." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:custom,cifar10,icn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [CIFAR10 dataset](https://www.tensorflow.org/datasets/catalog/cifar10) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use is built into TensorFlow. The trained model predicts which type of class an image is from ten classes: airplane, automobile, bird, cat, deer, dog, frog, horse, ship, truck." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,training,batch_prediction,xai" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create a custom model, with a training pipeline, from a Python script in a Google prebuilt Docker container using the Vertex client library, and then do a batch prediction with explanations on the uploaded model. You can alternatively create custom models using `gcloud` command-line tool or online using Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex custom job for training a model.\n", + "- Train the TensorFlow model.\n", + "- Retrieve and load the model artifacts.\n", + "- View the model evaluation.\n", + "- Set explanation parameters.\n", + "- Upload the model as a Vertex `Model` resource.\n", + "- Make a batch prediction with explanations." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_tensorflow" + }, + "outputs": [], + "source": [ + "if os.environ[\"IS_TESTING\"]:\n", + " ! pip3 install -U tensorflow" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_cv2" + }, + "outputs": [], + "source": [ + "if os.environ[\"IS_TESTING\"]:\n", + " ! pip3 install -U opencv-python" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:v1beta1,protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:v1beta1,protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "import google.cloud.aiplatform_v1beta1 as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for training and prediction.\n", + "\n", + "Set the variables `TRAIN_GPU/TRAIN_NGPU` and `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU.\n", + "\n", + "*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3 -- which is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n", + " TRAIN_GPU, TRAIN_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n", + " )\n", + "else:\n", + " TRAIN_GPU, TRAIN_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (None, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:training,prediction" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for training and prediction\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-3:latest`\n", + " - TensorFlow 2.4\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-4:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/training/xgboost-cpu.1-1`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/training/scikit-learn-cpu.0-23:latest`\n", + " - Pytorch\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-5:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-6:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-7:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for training](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers).\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-3:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-90:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-82:latest`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-23:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-22:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-20:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for prediction](https://cloud.google.com/vertex-ai/docs/predictions/pre-built-containers)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf2-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf2-cpu.{}\".format(TF)\n", + "else:\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "TRAIN_IMAGE = \"gcr.io/cloud-aiplatform/training/{}:latest\".format(TRAIN_VERSION)\n", + "DEPLOY_IMAGE = \"gcr.io/cloud-aiplatform/prediction/{}:latest\".format(DEPLOY_VERSION)\n", + "\n", + "print(\"Training:\", TRAIN_IMAGE, TRAIN_GPU, TRAIN_NGPU)\n", + "print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU, DEPLOY_NGPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:training,prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for training and prediction.\n", + "\n", + "- Set the variables `TRAIN_COMPUTE` and `DEPLOY_COMPUTE` to configure the compute resources for the VMs you will use for for training and prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: The following is not supported for training:*\n", + "\n", + " - `standard`: 2 vCPUs\n", + " - `highcpu`: 2, 4 and 8 vCPUs\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_TRAIN_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Train machine type\", TRAIN_COMPUTE)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own custom model and training for CIFAR10." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:custom" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Model Service for `Model` resources.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:custom" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"job\"] = create_job_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_model" + }, + "source": [ + "## Train a model\n", + "\n", + "There are two ways you can train a custom model using a container image:\n", + "\n", + "- **Use a Google Cloud prebuilt container**. If you use a prebuilt container, you will additionally specify a Python package to install into the container image. This Python package contains your code for training a custom model.\n", + "\n", + "- **Use your own custom container image**. If you use your own container, the container needs to contain your code for training a custom model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_specification:prebuilt_container" + }, + "source": [ + "## Prepare your custom job specification\n", + "\n", + "Now that your clients are ready, your first step is to create a Job Specification for your custom training job. The job specification will consist of the following:\n", + "\n", + "- `worker_pool_spec` : The specification of the type of machine(s) you will use for training and how many (single or distributed)\n", + "- `python_package_spec` : The specification of the Python package to be installed with the pre-built container." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "source": [ + "### Prepare your machine specification\n", + "\n", + "Now define the machine specification for your custom training job. This tells Vertex what type of machine instance to provision for the training.\n", + " - `machine_type`: The type of GCP instance to provision -- e.g., n1-standard-8.\n", + " - `accelerator_type`: The type, if any, of hardware accelerator. In this tutorial if you previously set the variable `TRAIN_GPU != None`, you are using a GPU; otherwise you will use a CPU.\n", + " - `accelerator_count`: The number of accelerators." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "outputs": [], + "source": [ + "if TRAIN_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": TRAIN_COMPUTE,\n", + " \"accelerator_type\": TRAIN_GPU,\n", + " \"accelerator_count\": TRAIN_NGPU,\n", + " }\n", + "else:\n", + " machine_spec = {\"machine_type\": TRAIN_COMPUTE, \"accelerator_count\": 0}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "source": [ + "### Prepare your disk specification\n", + "\n", + "(optional) Now define the disk specification for your custom training job. This tells Vertex what type and size of disk to provision in each machine instance for the training.\n", + "\n", + " - `boot_disk_type`: Either SSD or Standard. SSD is faster, and Standard is less expensive. Defaults to SSD.\n", + " - `boot_disk_size_gb`: Size of disk in GB." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "outputs": [], + "source": [ + "DISK_TYPE = \"pd-ssd\" # [ pd-ssd, pd-standard]\n", + "DISK_SIZE = 200 # GB\n", + "\n", + "disk_spec = {\"boot_disk_type\": DISK_TYPE, \"boot_disk_size_gb\": DISK_SIZE}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container" + }, + "source": [ + "### Define the worker pool specification\n", + "\n", + "Next, you define the worker pool specification for your custom training job. The worker pool specification will consist of the following:\n", + "\n", + "- `replica_count`: The number of instances to provision of this machine type.\n", + "- `machine_spec`: The hardware specification.\n", + "- `disk_spec` : (optional) The disk storage specification.\n", + "\n", + "- `python_package`: The Python training package to install on the VM instance(s) and which Python module to invoke, along with command line arguments for the Python module.\n", + "\n", + "Let's dive deeper now into the python package specification:\n", + "\n", + "-`executor_image_spec`: This is the docker image which is configured for your custom training job.\n", + "\n", + "-`package_uris`: This is a list of the locations (URIs) of your python training packages to install on the provisioned instance. The locations need to be in a Cloud Storage bucket. These can be either individual python files or a zip (archive) of an entire package. In the later case, the job service will unzip (unarchive) the contents into the docker image.\n", + "\n", + "-`python_module`: The Python module (script) to invoke for running the custom training job. In this example, you will be invoking `trainer.task.py` -- note that it was not neccessary to append the `.py` suffix.\n", + "\n", + "-`args`: The command line arguments to pass to the corresponding Pythom module. In this example, you will be setting:\n", + " - `\"--model-dir=\" + MODEL_DIR` : The Cloud Storage location where to store the model artifacts. There are two ways to tell the training script where to save the model artifacts:\n", + " - direct: You pass the Cloud Storage location as a command line argument to your training script (set variable `DIRECT = True`), or\n", + " - indirect: The service passes the Cloud Storage location as the environment variable `AIP_MODEL_DIR` to your training script (set variable `DIRECT = False`). In this case, you tell the service the model artifact location in the job specification.\n", + " - `\"--epochs=\" + EPOCHS`: The number of epochs for training.\n", + " - `\"--steps=\" + STEPS`: The number of steps (batches) per epoch.\n", + " - `\"--distribute=\" + TRAIN_STRATEGY\"` : The training distribution strategy to use for single or distributed training.\n", + " - `\"single\"`: single device.\n", + " - `\"mirror\"`: all GPU devices on a single compute instance.\n", + " - `\"multi\"`: all GPU devices on all compute instances." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_job_\" + TIMESTAMP\n", + "MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, JOB_NAME)\n", + "\n", + "if not TRAIN_NGPU or TRAIN_NGPU < 2:\n", + " TRAIN_STRATEGY = \"single\"\n", + "else:\n", + " TRAIN_STRATEGY = \"mirror\"\n", + "\n", + "EPOCHS = 20\n", + "STEPS = 100\n", + "\n", + "DIRECT = True\n", + "if DIRECT:\n", + " CMDARGS = [\n", + " \"--model-dir=\" + MODEL_DIR,\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "else:\n", + " CMDARGS = [\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "\n", + "worker_pool_spec = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": machine_spec,\n", + " \"disk_spec\": disk_spec,\n", + " \"python_package_spec\": {\n", + " \"executor_image_uri\": TRAIN_IMAGE,\n", + " \"package_uris\": [BUCKET_NAME + \"/trainer_cifar10.tar.gz\"],\n", + " \"python_module\": \"trainer.task\",\n", + " \"args\": CMDARGS,\n", + " },\n", + " }\n", + "]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "assemble_custom_job_specification" + }, + "source": [ + "### Assemble a job specification\n", + "\n", + "Now assemble the complete description for the custom job specification:\n", + "\n", + "- `display_name`: The human readable name you assign to this custom job.\n", + "- `job_spec`: The specification for the custom job.\n", + " - `worker_pool_specs`: The specification for the machine VM instances.\n", + " - `base_output_directory`: This tells the service the Cloud Storage location where to save the model artifacts (when variable `DIRECT = False`). The service will then pass the location to the training script as the environment variable `AIP_MODEL_DIR`, and the path will be of the form:\n", + "\n", + " /model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "assemble_custom_job_specification" + }, + "outputs": [], + "source": [ + "if DIRECT:\n", + " job_spec = {\"worker_pool_specs\": worker_pool_spec}\n", + "else:\n", + " job_spec = {\n", + " \"worker_pool_specs\": worker_pool_spec,\n", + " \"base_output_directory\": {\"output_uri_prefix\": MODEL_DIR},\n", + " }\n", + "\n", + "custom_job = {\"display_name\": JOB_NAME, \"job_spec\": job_spec}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "examine_training_package" + }, + "source": [ + "### Examine the training package\n", + "\n", + "#### Package layout\n", + "\n", + "Before you start the training, you will look at how a Python package is assembled for a custom training job. When unarchived, the package contains the following directory/file layout.\n", + "\n", + "- PKG-INFO\n", + "- README.md\n", + "- setup.cfg\n", + "- setup.py\n", + "- trainer\n", + " - \\_\\_init\\_\\_.py\n", + " - task.py\n", + "\n", + "The files `setup.cfg` and `setup.py` are the instructions for installing the package into the operating environment of the Docker image.\n", + "\n", + "The file `trainer/task.py` is the Python script for executing the custom training job. *Note*, when we referred to it in the worker pool specification, we replace the directory slash with a dot (`trainer.task`) and dropped the file suffix (`.py`).\n", + "\n", + "#### Package Assembly\n", + "\n", + "In the following cells, you will assemble the training package." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "examine_training_package" + }, + "outputs": [], + "source": [ + "# Make folder for Python training script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\ntag_build =\\n\\ntag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'tensorflow_datasets==1.3.0',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\nName: CIFAR10 image classification\\n\\nVersion: 0.0.0\\n\\nSummary: Demostration training script\\n\\nHome-page: www.google.com\\n\\nAuthor: Google\\n\\nAuthor-email: aferlitsch@google.com\\n\\nLicense: Public\\n\\nDescription: Demo\\n\\nPlatform: Vertex\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:cifar10" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the training script task.py. We won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n", + "- Loads CIFAR10 dataset from TF Datasets (tfds).\n", + "- Builds a model using TF.Keras model API.\n", + "- Compiles the model (`compile()`).\n", + "- Sets a training distribution strategy according to the argument `args.distribute`.\n", + "- Trains the model (`fit()`) with epochs and steps according to the arguments `args.epochs` and `args.steps`\n", + "- Saves the trained model (`save(args.model_dir)`) to the specified model directory." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:cifar10" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Single, Mirror and Multi-Machine Distributed Training for CIFAR-10\n", + "\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "import argparse\n", + "import os\n", + "import sys\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv(\"AIP_MODEL_DIR\"), type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.01, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=10, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=200, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "print('DEVICES', device_lib.list_local_devices())\n", + "\n", + "# Single Machine, single compute device\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "# Single Machine, multiple compute device\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "# Multiple Machine, multiple compute device\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "# Multi-worker configuration\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "# Preparing dataset\n", + "BUFFER_SIZE = 10000\n", + "BATCH_SIZE = 64\n", + "\n", + "\n", + "def make_datasets_unbatched():\n", + "\n", + " # Scaling CIFAR10 data from (0, 255] to (0., 1.]\n", + " def scale(image, label):\n", + " image = tf.cast(image, tf.float32)\n", + " image /= 255.0\n", + " return image, label\n", + "\n", + "\n", + " datasets, info = tfds.load(name='cifar10',\n", + " with_info=True,\n", + " as_supervised=True)\n", + " return datasets['train'].map(scale).cache().shuffle(BUFFER_SIZE).repeat()\n", + "\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_cnn_model():\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu', input_shape=(32, 32, 3)),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu'),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Flatten(),\n", + " tf.keras.layers.Dense(10, activation='softmax')\n", + " ])\n", + " model.compile(\n", + " loss=tf.keras.losses.sparse_categorical_crossentropy,\n", + " optimizer=tf.keras.optimizers.SGD(learning_rate=args.lr),\n", + " metrics=['accuracy'])\n", + " return model\n", + "\n", + "\n", + "# Train the model\n", + "NUM_WORKERS = strategy.num_replicas_in_sync\n", + "# Here the batch size scales up by number of workers since\n", + "# `tf.data.Dataset.batch` expects the global batch size.\n", + "GLOBAL_BATCH_SIZE = BATCH_SIZE * NUM_WORKERS\n", + "train_dataset = make_datasets_unbatched().batch(GLOBAL_BATCH_SIZE)\n", + "\n", + "with strategy.scope():\n", + " # Creation of dataset, and model building/compiling need to be within\n", + " # `strategy.scope()`.\n", + " model = build_and_compile_cnn_model()\n", + "\n", + "model.fit(x=train_dataset, epochs=args.epochs, steps_per_epoch=args.steps)\n", + "model.save(args.model_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tarball_training_script" + }, + "source": [ + "#### Store training script on your Cloud Storage bucket\n", + "\n", + "Next, you package the training folder into a compressed tar ball, and then store it in your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tarball_training_script" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_cifar10.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job" + }, + "source": [ + "### Train the model\n", + "\n", + "\n", + "Now start the training of your custom training job on Vertex. Use this helper function `create_custom_job`, which takes the following parameter:\n", + "\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "The helper function calls job client service's `create_custom_job` method, with the following parameters:\n", + "\n", + "-`parent`: The Vertex location path to `Dataset`, `Model` and `Endpoint` resources.\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "You will display a handful of the fields returned in `response` object, with the two that are of most interest are:\n", + "\n", + "`response.name`: The Vertex fully qualified identifier assigned to this custom training job. You save this identifier for using in subsequent steps.\n", + "\n", + "`response.state`: The current state of the custom training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job" + }, + "outputs": [], + "source": [ + "def create_custom_job(custom_job):\n", + " response = clients[\"job\"].create_custom_job(parent=PARENT, custom_job=custom_job)\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = create_custom_job(custom_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "job_id:response" + }, + "source": [ + "Now get the unique identifier for the custom job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom job\n", + "job_id = response.name\n", + "# The short numeric ID for the custom job\n", + "job_short_id = job_id.split(\"/\")[-1]\n", + "\n", + "print(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_custom_job" + }, + "source": [ + "### Get information on a custom job\n", + "\n", + "Next, use this helper function `get_custom_job`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "The helper function calls the job client service's`get_custom_job` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "If you recall, you got the Vertex fully qualified identifier for the custom job in the `response.name` field when you called the `create_custom_job` method, and saved the identifier in the variable `job_id`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_custom_job" + }, + "outputs": [], + "source": [ + "def get_custom_job(name, silent=False):\n", + " response = clients[\"job\"].get_custom_job(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = get_custom_job(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete:custom" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, we will need to know the location of the saved model, which the Python script saved in your local Cloud Storage bucket at `MODEL_DIR + '/saved_model.pb'`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete:custom" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_custom_job(job_id, True)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_path_to_deploy = None\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " if not DIRECT:\n", + " MODEL_DIR = MODEL_DIR + \"/model\"\n", + " model_path_to_deploy = MODEL_DIR\n", + " print(\"Training Time:\", response.update_time - response.create_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model_to_deploy:\", model_path_to_deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "load_saved_model" + }, + "source": [ + "## Load the saved model\n", + "\n", + "Your model is stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Now load it from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n", + "\n", + "To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "load_saved_model" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:image" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load the CIFAR10 test (holdout) data from `tf.keras.datasets`, using the method `load_data()`. This will return the dataset as a tuple of two elements. The first element is the training data and the second is the test data. Each element is also a tuple of two elements: the image data, and the corresponding labels.\n", + "\n", + "You don't need the training data, and hence why we loaded it as `(_, _)`.\n", + "\n", + "Before you can run the data through evaluation, you need to preprocess it:\n", + "\n", + "x_test:\n", + "1. Normalize (rescaling) the pixel data by dividing each pixel by 255. This will replace each single byte integer pixel with a 32-bit floating point number between 0 and 1.\n", + "\n", + "y_test:
\n", + "2. The labels are currently scalar (sparse). If you look back at the `compile()` step in the `trainer/task.py` script, you will find that it was compiled for sparse labels. So we don't need to do anything more." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "evaluate_custom_model:image,cifar10" + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "from tensorflow.keras.datasets import cifar10\n", + "\n", + "(_, _), (x_test, y_test) = cifar10.load_data()\n", + "x_test = (x_test / 255.0).astype(np.float32)\n", + "\n", + "print(x_test.shape, y_test.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "perform_evaluation_custom" + }, + "source": [ + "### Perform the model evaluation\n", + "\n", + "Now evaluate how well the model in the custom job did." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "perform_evaluation_custom" + }, + "outputs": [], + "source": [ + "model.evaluate(x_test, y_test)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "how_serving_function_works" + }, + "source": [ + "## Upload the model for serving\n", + "\n", + "Next, you will upload your TF.Keras model from the custom job to Vertex `Model` service, which will create a Vertex `Model` resource for your custom model. During upload, you need to define a serving function to convert data to the format your model expects. If you send encoded data to Vertex, your serving function ensures that the data is decoded on the model server before it is passed as input to your model.\n", + "\n", + "### How does the serving function work\n", + "\n", + "When you send a request to an online prediction server, the request is received by a HTTP server. The HTTP server extracts the prediction request from the HTTP request content body. The extracted prediction request is forwarded to the serving function. For Google pre-built prediction containers, the request content is passed to the serving function as a `tf.string`.\n", + "\n", + "The serving function consists of two parts:\n", + "\n", + "- `preprocessing function`:\n", + " - Converts the input (`tf.string`) to the input shape and data type of the underlying model (dynamic graph).\n", + " - Performs the same preprocessing of the data that was done during training the underlying model -- e.g., normalizing, scaling, etc.\n", + "- `post-processing function`:\n", + " - Converts the model output to format expected by the receiving application -- e.q., compresses the output.\n", + " - Packages the output for the the receiving application -- e.g., add headings, make JSON object, etc.\n", + "\n", + "Both the preprocessing and post-processing functions are converted to static graphs which are fused to the model. The output from the underlying model is passed to the post-processing function. The post-processing function passes the converted/packaged output back to the HTTP server. The HTTP server returns the output as the HTTP response content.\n", + "\n", + "One consideration you need to consider when building serving functions for TF.Keras models is that they run as static graphs. That means, you cannot use TF graph operations that require a dynamic graph. If you do, you will get an error during the compile of the serving function which will indicate that you are using an EagerTensor which is not supported." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_image:xai" + }, + "source": [ + "### Serving function for image data\n", + "\n", + "To pass images to the prediction service, you encode the compressed (e.g., JPEG) image bytes into base 64 -- which makes the content safe from modification while transmitting binary data over the network. Since this deployed model expects input data as raw (uncompressed) bytes, you need to ensure that the base 64 encoded data gets converted back to raw bytes before it is passed as input to the deployed model.\n", + "\n", + "To resolve this, define a serving function (`serving_fn`) and attach it to the model as a preprocessing step. Add a `@tf.function` decorator so the serving function is fused to the underlying model (instead of upstream on a CPU).\n", + "\n", + "When you send a prediction or explanation request, the content of the request is base 64 decoded into a Tensorflow string (`tf.string`), which is passed to the serving function (`serving_fn`). The serving function preprocesses the `tf.string` into raw (uncompressed) numpy bytes (`preprocess_fn`) to match the input requirements of the model:\n", + "- `io.decode_jpeg`- Decompresses the JPG image which is returned as a Tensorflow tensor with three channels (RGB).\n", + "- `image.convert_image_dtype` - Changes integer pixel values to float 32.\n", + "- `image.resize` - Resizes the image to match the input shape for the model.\n", + "- `resized / 255.0` - Rescales (normalization) the pixel data between 0 and 1.\n", + "\n", + "At this point, the data can be passed to the model (`m_call`).\n", + "\n", + "#### XAI Signatures\n", + "\n", + "When the serving function is saved back with the underlying model (`tf.saved_model.save`), you specify the input layer of the serving function as the signature `serving_default`.\n", + "\n", + "For XAI image models, you need to save two additional signatures from the serving function:\n", + "\n", + "- `xai_preprocess`: The preprocessing function in the serving function.\n", + "- `xai_model`: The concrete function for calling the model." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_image:xai" + }, + "outputs": [], + "source": [ + "CONCRETE_INPUT = \"numpy_inputs\"\n", + "\n", + "\n", + "def _preprocess(bytes_input):\n", + " decoded = tf.io.decode_jpeg(bytes_input, channels=3)\n", + " decoded = tf.image.convert_image_dtype(decoded, tf.float32)\n", + " resized = tf.image.resize(decoded, size=(32, 32))\n", + " rescale = tf.cast(resized / 255.0, tf.float32)\n", + " return rescale\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def preprocess_fn(bytes_inputs):\n", + " decoded_images = tf.map_fn(\n", + " _preprocess, bytes_inputs, dtype=tf.float32, back_prop=False\n", + " )\n", + " return {\n", + " CONCRETE_INPUT: decoded_images\n", + " } # User needs to make sure the key matches model's input\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def serving_fn(bytes_inputs):\n", + " images = preprocess_fn(bytes_inputs)\n", + " prob = m_call(**images)\n", + " return prob\n", + "\n", + "\n", + "m_call = tf.function(model.call).get_concrete_function(\n", + " [tf.TensorSpec(shape=[None, 32, 32, 3], dtype=tf.float32, name=CONCRETE_INPUT)]\n", + ")\n", + "\n", + "tf.saved_model.save(\n", + " model,\n", + " model_path_to_deploy,\n", + " signatures={\n", + " \"serving_default\": serving_fn,\n", + " # Required for XAI\n", + " \"xai_preprocess\": preprocess_fn,\n", + " \"xai_model\": m_call,\n", + " },\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_signature:xai" + }, + "source": [ + "## Get the serving function signature\n", + "\n", + "You can get the signatures of your model's input and output layers by reloading the model into memory, and querying it for the signatures corresponding to each layer.\n", + "\n", + "When making a prediction request, you need to route the request to the serving function instead of the model, so you need to know the input layer name of the serving function -- which you will use later when you make a prediction request.\n", + "\n", + "You also need to know the name of the serving function's input and output layer for constructing the explanation metadata -- which is discussed subsequently." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_signature:xai" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_path_to_deploy)\n", + "\n", + "serving_input = list(\n", + " loaded.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "print(\"Serving function input:\", serving_input)\n", + "serving_output = list(loaded.signatures[\"serving_default\"].structured_outputs.keys())[0]\n", + "print(\"Serving function output:\", serving_output)\n", + "\n", + "input_name = model.input.name\n", + "print(\"Model input name:\", input_name)\n", + "output_name = model.output.name\n", + "print(\"Model output name:\", output_name)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "explanation_spec" + }, + "source": [ + "### Explanation Specification\n", + "\n", + "To get explanations when doing a prediction, you must enable the explanation capability and set corresponding settings when you upload your custom model to an Vertex `Model` resource. These settings are referred to as the explanation metadata, which consists of:\n", + "\n", + "- `parameters`: This is the specification for the explainability algorithm to use for explanations on your model. You can choose between:\n", + " - Shapley - *Note*, not recommended for image data -- can be very long running\n", + " - XRAI\n", + " - Integrated Gradients\n", + "- `metadata`: This is the specification for how the algoithm is applied on your custom model.\n", + "\n", + "#### Explanation Parameters\n", + "\n", + "Let's first dive deeper into the settings for the explainability algorithm.\n", + "\n", + "#### Shapley\n", + "\n", + "Assigns credit for the outcome to each feature, and considers different permutations of the features. This method provides a sampling approximation of exact Shapley values.\n", + "\n", + "Use Cases:\n", + " - Classification and regression on tabular data.\n", + "\n", + "Parameters:\n", + "\n", + "- `path_count`: This is the number of paths over the features that will be processed by the algorithm. An exact approximation of the Shapley values requires M! paths, where M is the number of features. For the CIFAR10 dataset, this would be 784 (28*28).\n", + "\n", + "For any non-trival number of features, this is too compute expensive. You can reduce the number of paths over the features to M * `path_count`.\n", + "\n", + "#### Integrated Gradients\n", + "\n", + "A gradients-based method to efficiently compute feature attributions with the same axiomatic properties as the Shapley value.\n", + "\n", + "Use Cases:\n", + " - Classification and regression on tabular data.\n", + " - Classification on image data.\n", + "\n", + "Parameters:\n", + "\n", + "- `step_count`: This is the number of steps to approximate the remaining sum. The more steps, the more accurate the integral approximation. The general rule of thumb is 50 steps, but as you increase so does the compute time.\n", + "\n", + "#### XRAI\n", + "\n", + "Based on the integrated gradients method, XRAI assesses overlapping regions of the image to create a saliency map, which highlights relevant regions of the image rather than pixels.\n", + "\n", + "Use Cases:\n", + "\n", + " - Classification on image data.\n", + "\n", + "Parameters:\n", + "\n", + "- `step_count`: This is the number of steps to approximate the remaining sum. The more steps, the more accurate the integral approximation. The general rule of thumb is 50 steps, but as you increase so does the compute time.\n", + "\n", + "In the next code cell, set the variable `XAI` to which explainabilty algorithm you will use on your custom model." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "explanation_spec" + }, + "outputs": [], + "source": [ + "XAI = \"ig\" # [ shapley, ig, xrai ]\n", + "\n", + "if XAI == \"shapley\":\n", + " PARAMETERS = {\"sampled_shapley_attribution\": {\"path_count\": 10}}\n", + "elif XAI == \"ig\":\n", + " PARAMETERS = {\"integrated_gradients_attribution\": {\"step_count\": 50}}\n", + "elif XAI == \"xrai\":\n", + " PARAMETERS = {\"xrai_attribution\": {\"step_count\": 50}}\n", + "\n", + "parameters = aip.ExplanationParameters(PARAMETERS)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "explanation_metadata:image" + }, + "source": [ + "#### Explanation Metadata\n", + "\n", + "Let's first dive deeper into the explanation metadata, which consists of:\n", + "\n", + "- `outputs`: A scalar value in the output to attribute -- what to explain. For example, in a probability output \\[0.1, 0.2, 0.7\\] for classification, one wants an explanation for 0.7. Consider the following formulae, where the output is `y` and that is what we want to explain.\n", + "\n", + " y = f(x)\n", + "\n", + "Consider the following formulae, where the outputs are `y` and `z`. Since we can only do attribution for one scalar value, we have to pick whether we want to explain the output `y` or `z`. Assume in this example the model is object detection and y and z are the bounding box and the object classification. You would want to pick which of the two outputs to explain.\n", + "\n", + " y, z = f(x)\n", + "\n", + "The dictionary format for `outputs` is:\n", + "\n", + " { \"outputs\": { \"[your_display_name]\":\n", + " \"output_tensor_name\": [layer]\n", + " }\n", + " }\n", + "\n", + "
\n", + " - [your_display_name]: A human readable name you assign to the output to explain. A common example is \"probability\".
\n", + " - \"output_tensor_name\": The key/value field to identify the output layer to explain.
\n", + " - [layer]: The output layer to explain. In a single task model, like a tabular regressor, it is the last (topmost) layer in the model.\n", + "
\n", + "\n", + "- `inputs`: The features for attribution -- how they contributed to the output. Consider the following formulae, where `a` and `b` are the features. We have to pick which features to explain how the contributed. Assume that this model is deployed for A/B testing, where `a` are the data_items for the prediction and `b` identifies whether the model instance is A or B. You would want to pick `a` (or some subset of) for the features, and not `b` since it does not contribute to the prediction.\n", + "\n", + " y = f(a,b)\n", + "\n", + "The minimum dictionary format for `inputs` is:\n", + "\n", + " { \"inputs\": { \"[your_display_name]\":\n", + " \"input_tensor_name\": [layer]\n", + " }\n", + " }\n", + "\n", + "
\n", + " - [your_display_name]: A human readable name you assign to the input to explain. A common example is \"features\".
\n", + " - \"input_tensor_name\": The key/value field to identify the input layer for the feature attribution.
\n", + " - [layer]: The input layer for feature attribution. In a single input tensor model, it is the first (bottom-most) layer in the model.\n", + "
\n", + "\n", + "Since the inputs to the model are tabular, you can specify the following two additional fields as reporting/visualization aids:\n", + "\n", + "
\n", + " - \"modality\": \"image\": Indicates the field values are image data.\n", + "
\n", + "\n", + "Since the inputs to the model are images, you can specify the following additional fields as reporting/visualization aids:\n", + "\n", + "
\n", + " - \"modality\": \"image\": Indicates the field values are image data.\n", + "
" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "explanation_metadata:image" + }, + "outputs": [], + "source": [ + "random_baseline = np.random.rand(32, 32, 3)\n", + "input_baselines = [{\"number_vaue\": x} for x in random_baseline]\n", + "\n", + "INPUT_METADATA = {\"input_tensor_name\": CONCRETE_INPUT, \"modality\": \"image\"}\n", + "\n", + "OUTPUT_METADATA = {\"output_tensor_name\": serving_output}\n", + "\n", + "input_metadata = aip.ExplanationMetadata.InputMetadata(INPUT_METADATA)\n", + "output_metadata = aip.ExplanationMetadata.OutputMetadata(OUTPUT_METADATA)\n", + "\n", + "metadata = aip.ExplanationMetadata(\n", + " inputs={\"image\": input_metadata}, outputs={\"class\": output_metadata}\n", + ")\n", + "\n", + "explanation_spec = aip.ExplanationSpec(metadata=metadata, parameters=parameters)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "upload_the_model:explanation" + }, + "source": [ + "### Upload the model\n", + "\n", + "Use this helper function `upload_model` to upload your model, stored in SavedModel format, up to the `Model` service, which will instantiate a Vertex `Model` resource instance for your model. Once you've done that, you can use the `Model` resource instance in the same way as any other Vertex `Model` resource instance, such as deploying to an `Endpoint` resource for serving predictions.\n", + "\n", + "The helper function takes the following parameters:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` service.\n", + "- `image_uri`: The container image for the model deployment.\n", + "- `model_uri`: The Cloud Storage path to our SavedModel artificat. For this tutorial, this is the Cloud Storage location where the `trainer/task.py` saved the model artifacts, which we specified in the variable `MODEL_DIR`.\n", + "\n", + "The helper function calls the `Model` client service's method `upload_model`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for `Dataset`, `Model` and `Endpoint` resources.\n", + "- `model`: The specification for the Vertex `Model` resource instance.\n", + "\n", + "Let's now dive deeper into the Vertex model specification `model`. This is a dictionary object that consists of the following fields:\n", + "\n", + "- `display_name`: A human readable name for the `Model` resource.\n", + "- `metadata_schema_uri`: Since your model was built without an Vertex `Dataset` resource, you will leave this blank (`''`).\n", + "- `artificat_uri`: The Cloud Storage path where the model is stored in SavedModel format.\n", + "- `container_spec`: This is the specification for the Docker container that will be installed on the `Endpoint` resource, from which the `Model` resource will serve predictions. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + "- `explanation_spec`: This is the specification for enabling explainability for your model.\n", + "\n", + "Uploading a model into a Vertex Model resource returns a long running operation, since it may take a few moments. You call response.result(), which is a synchronous call and will return when the Vertex Model resource is ready.\n", + "\n", + "The helper function returns the Vertex fully qualified identifier for the corresponding Vertex Model instance upload_model_response.model. You will save the identifier for subsequent steps in the variable model_to_deploy_id." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "upload_the_model:explanation" + }, + "outputs": [], + "source": [ + "IMAGE_URI = DEPLOY_IMAGE\n", + "\n", + "\n", + "def upload_model(display_name, image_uri, model_uri):\n", + "\n", + " model = aip.Model(\n", + " display_name=display_name,\n", + " artifact_uri=model_uri,\n", + " metadata_schema_uri=\"\",\n", + " explanation_spec=explanation_spec,\n", + " container_spec={\"image_uri\": image_uri},\n", + " )\n", + "\n", + " response = clients[\"model\"].upload_model(parent=PARENT, model=model)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " upload_model_response = response.result(timeout=180)\n", + " print(\"upload_model_response\")\n", + " print(\" model:\", upload_model_response.model)\n", + " return upload_model_response.model\n", + "\n", + "\n", + "model_to_deploy_id = upload_model(\n", + " \"cifar10-\" + TIMESTAMP, IMAGE_URI, model_path_to_deploy\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_model" + }, + "source": [ + "### Get `Model` resource information\n", + "\n", + "Now let's get the model information for just your model. Use this helper function `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource.\n", + "\n", + "This helper function calls the Vertex `Model` client service's method `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_model" + }, + "outputs": [], + "source": [ + "def get_model(name):\n", + " response = clients[\"model\"].get_model(name=name)\n", + " print(response)\n", + "\n", + "\n", + "get_model(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy:batch_prediction" + }, + "source": [ + "## Model deployment for batch prediction\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created for batch prediction. This differs from deploying a `Model` resource for on-demand prediction.\n", + "\n", + "For online prediction, you:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource.\n", + "\n", + "3. Make online prediction requests to the `Endpoint` resource.\n", + "\n", + "For batch-prediction, you:\n", + "\n", + "1. Create a batch prediction job.\n", + "\n", + "2. The job service will provision resources for the batch prediction request.\n", + "\n", + "3. The results of the batch prediction request are returned to the caller.\n", + "\n", + "4. The job service will unprovision the resoures for the batch prediction request." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a batch prediction request\n", + "\n", + "Now do a batch prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_items:test" + }, + "source": [ + "### Get test items\n", + "\n", + "You will use examples out of the test (holdout) portion of the dataset as a test items." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_items:test" + }, + "outputs": [], + "source": [ + "test_image_1 = x_test[0]\n", + "test_label_1 = y_test[0]\n", + "test_image_2 = x_test[1]\n", + "test_label_2 = y_test[1]\n", + "print(test_image_1.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "prepare_test_items:test,image" + }, + "source": [ + "### Prepare the request content\n", + "You are going to send the CIFAR10 images as compressed JPG image, instead of the raw uncompressed bytes:\n", + "\n", + "- `cv2.imwrite`: Use openCV to write the uncompressed image to disk as a compressed JPEG image.\n", + " - Denormalize the image data from \\[0,1) range back to [0,255).\n", + " - Convert the 32-bit floating point values to 8-bit unsigned integers." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "prepare_test_items:test,image" + }, + "outputs": [], + "source": [ + "import cv2\n", + "\n", + "cv2.imwrite(\"tmp1.jpg\", (test_image_1 * 255).astype(np.uint8))\n", + "cv2.imwrite(\"tmp2.jpg\", (test_image_2 * 255).astype(np.uint8))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "copy_test_items:test" + }, + "source": [ + "### Copy test item(s)\n", + "\n", + "For the batch prediction, you will copy the test items over to your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copy_test_items:test" + }, + "outputs": [], + "source": [ + "! gsutil cp tmp1.jpg $BUCKET_NAME/tmp1.jpg\n", + "! gsutil cp tmp2.jpg $BUCKET_NAME/tmp2.jpg\n", + "\n", + "test_item_1 = BUCKET_NAME + \"/tmp1.jpg\"\n", + "test_item_2 = BUCKET_NAME + \"/tmp2.jpg\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:custom,image" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "Now make a batch input file, which you will store in your local Cloud Storage bucket. The batch input file can only be in JSONL format. For JSONL file, you make one dictionary entry per line for each data item (instance). The dictionary contains the key/value pairs:\n", + "\n", + "- `input_name`: the name of the input layer of the underlying model.\n", + "- `'b64'`: A key that indicates the content is base64 encoded.\n", + "- `content`: The compressed JPG image bytes as a base64 encoded string.\n", + "\n", + "Each instance in the prediction request is a dictionary entry of the form:\n", + "\n", + " {serving_input: {'b64': content}}\n", + "\n", + "To pass the image data to the prediction service you encode the bytes into base64 -- which makes the content safe from modification when transmitting binary data over the network.\n", + "\n", + "- `tf.io.read_file`: Read the compressed JPG images into memory as raw bytes.\n", + "- `base64.b64encode`: Encode the raw bytes into a base64 encoded string." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_file:custom,image" + }, + "outputs": [], + "source": [ + "import base64\n", + "import json\n", + "\n", + "gcs_input_uri = BUCKET_NAME + \"/\" + \"test.jsonl\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " bytes = tf.io.read_file(test_item_1)\n", + " b64str = base64.b64encode(bytes.numpy()).decode(\"utf-8\")\n", + " data = {serving_input: {\"b64\": b64str}}\n", + " f.write(json.dumps(data) + \"\\n\")\n", + " bytes = tf.io.read_file(test_item_2)\n", + " b64str = base64.b64encode(bytes.numpy()).decode(\"utf-8\")\n", + " data = {serving_input: {\"b64\": b64str}}\n", + " f.write(json.dumps(data) + \"\\n\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your batch prediction requests:\n", + "\n", + "- Single Instance: The batch prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The batch prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and batch prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The batch prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_request:custom" + }, + "source": [ + "### Make batch prediction request\n", + "\n", + "Now that your batch of two test items is ready, let's do the batch request. Use this helper function `create_batch_prediction_job`, with the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the prediction job.\n", + "- `model_name`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `gcs_source_uri`: The Cloud Storage path to the input file -- which you created above.\n", + "- `gcs_destination_output_uri_prefix`: The Cloud Storage path that the service will write the predictions to.\n", + "- `parameters`: Additional filtering parameters for serving prediction results.\n", + "\n", + "The helper function calls the job client service's `create_batch_prediction_job` metho, with the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for Dataset, Model and Pipeline resources.\n", + "- `batch_prediction_job`: The specification for the batch prediction job.\n", + "\n", + "Let's now dive into the specification for the `batch_prediction_job`:\n", + "\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `dedicated_resources`: The compute resources to provision for the batch prediction job.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `starting_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "- `model_parameters`: Additional filtering parameters for serving prediction results. No Additional parameters are supported for custom models.\n", + "- `input_config`: The input source and format type for the instances to predict.\n", + " - `instances_format`: The format of the batch prediction request file: `csv` or `jsonl`.\n", + " - `gcs_source`: A list of one or more Cloud Storage paths to your batch prediction requests.\n", + "- `output_config`: The output destination and format for the predictions.\n", + " - `prediction_format`: The format of the batch prediction response file: `csv` or `jsonl`.\n", + " - `gcs_destination`: The output destination for the predictions.\n", + "\n", + "This call is an asychronous operation. You will print from the response object a few select fields, including:\n", + "\n", + "- `name`: The Vertex fully qualified identifier assigned to the batch prediction job.\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the Model resource.\n", + "- `generate_explanations`: Whether True/False explanations were provided with the predictions (explainability).\n", + "- `state`: The state of the prediction job (pending, running, etc).\n", + "\n", + "Since this call will take a few moments to execute, you will likely get `JobState.JOB_STATE_PENDING` for `state`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_request:custom" + }, + "outputs": [], + "source": [ + "BATCH_MODEL = \"cifar10_batch-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_batch_prediction_job(\n", + " display_name,\n", + " model_name,\n", + " gcs_source_uri,\n", + " gcs_destination_output_uri_prefix,\n", + " parameters=None,\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " batch_prediction_job = {\n", + " \"display_name\": display_name,\n", + " # Format: 'projects/{project}/locations/{location}/models/{model_id}'\n", + " \"model\": model_name,\n", + " \"model_parameters\": json_format.ParseDict(parameters, Value()),\n", + " \"input_config\": {\n", + " \"instances_format\": IN_FORMAT,\n", + " \"gcs_source\": {\"uris\": [gcs_source_uri]},\n", + " },\n", + " \"output_config\": {\n", + " \"predictions_format\": OUT_FORMAT,\n", + " \"gcs_destination\": {\"output_uri_prefix\": gcs_destination_output_uri_prefix},\n", + " },\n", + " \"dedicated_resources\": {\n", + " \"machine_spec\": machine_spec,\n", + " \"starting_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " },\n", + " \"generate_explanation\": True,\n", + " }\n", + " response = clients[\"job\"].create_batch_prediction_job(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + " )\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try:\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", response.labels)\n", + " return response\n", + "\n", + "\n", + "IN_FORMAT = \"jsonl\"\n", + "OUT_FORMAT = \"jsonl\"\n", + "\n", + "response = create_batch_prediction_job(\n", + " BATCH_MODEL, model_to_deploy_id, gcs_input_uri, BUCKET_NAME\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batch_job_id:response" + }, + "source": [ + "Now get the unique identifier for the batch prediction job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batch_job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the batch job\n", + "batch_job_id = response.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split(\"/\")[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_batch_prediction_job" + }, + "source": [ + "### Get information on a batch prediction job\n", + "\n", + "Use this helper function `get_batch_prediction_job`, with the following paramter:\n", + "\n", + "- `job_name`: The Vertex fully qualified identifier for the batch prediction job.\n", + "\n", + "The helper function calls the job client service's `get_batch_prediction_job` method, with the following paramter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the batch prediction job. In this tutorial, you will pass it the Vertex fully qualified identifier for your batch prediction job -- `batch_job_id`\n", + "\n", + "The helper function will return the Cloud Storage path to where the predictions are stored -- `gcs_destination`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_batch_prediction_job" + }, + "outputs": [], + "source": [ + "def get_batch_prediction_job(job_name, silent=False):\n", + " response = clients[\"job\"].get_batch_prediction_job(name=job_name)\n", + " if silent:\n", + " return response.output_config.gcs_destination.output_uri_prefix, response.state\n", + "\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try: # not all data types support explanations\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" error:\", response.error)\n", + " gcs_destination = response.output_config.gcs_destination\n", + " print(\" gcs_destination\")\n", + " print(\" output_uri_prefix:\", gcs_destination.output_uri_prefix)\n", + " return gcs_destination.output_uri_prefix, response.state\n", + "\n", + "\n", + "predictions, state = get_batch_prediction_job(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_the_explanations:custom,image" + }, + "source": [ + "### Get the predictions\n", + "\n", + "When the batch prediction is done processing, the job state will be `JOB_STATE_SUCCEEDED`.\n", + "\n", + "Finally you view the predictions stored at the Cloud Storage path you set as output. The predictions will be in a JSONL format, which you indicated at the time you made the batch prediction job, under a subfolder starting with the name `prediction`, and under that folder will be a file called `prediction.results-xxxxx-of-xxxxx`.\n", + "\n", + "Now display (cat) the contents. You will see multiple JSON objects, one for each prediction.\n", + "\n", + "Finally you view the explanations stored at the Cloud Storage path you set as output. The explanations will be in a JSONL format, which you indicated at the time you made the batch explanation job, under a subfolder starting with the name `prediction`, and under that folder will be a file called `explanation-results-xxxx-of-xxxx`.\n", + "\n", + "Let's display (cat) the contents. You will a row for each prediction -- in this case, there is just one row. The row is the softmax probability distribution for the corresponding CIFAR10 classes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_the_explanations:custom,image" + }, + "outputs": [], + "source": [ + "def get_latest_predictions(gcs_out_dir):\n", + " \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n", + " folders = !gsutil ls $gcs_out_dir\n", + " latest = \"\"\n", + " for folder in folders:\n", + " subfolder = folder.split(\"/\")[-2]\n", + " if subfolder.startswith(\"prediction-\"):\n", + " if subfolder > latest:\n", + " latest = folder[:-1]\n", + " return latest\n", + "\n", + "\n", + "while True:\n", + " predictions, state = get_batch_prediction_job(batch_job_id, True)\n", + " if state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", state)\n", + " if state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " folder = get_latest_predictions(predictions)\n", + " ! gsutil ls $folder/explanation.results*\n", + "\n", + " print(\"Results:\")\n", + " ! gsutil cat $folder/explanation.results*\n", + "\n", + " print(\"Errors:\")\n", + " ! gsutil cat $folder/prediction.errors*\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_custom_image_classification_batch_explain.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_custom_image_classification_online.ipynb b/notebooks/community/gapic/custom/showcase_custom_image_classification_online.ipynb new file mode 100644 index 000000000..3567fe6c4 --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_custom_image_classification_online.ipynb @@ -0,0 +1,2192 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Custom training image classification model for online prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to train and deploy a custom image classification model for online prediction." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:custom,cifar10,icn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [CIFAR10 dataset](https://www.tensorflow.org/datasets/catalog/cifar10) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use is built into TensorFlow. The trained model predicts which type of class an image is from ten classes: airplane, automobile, bird, cat, deer, dog, frog, horse, ship, truck." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,training,online_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create a custom model from a Python script in a Google prebuilt Docker container using the Vertex client library, and then do a prediction on the deployed model by sending data. You can alternatively create custom models using `gcloud` command-line tool or online using Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex custom job for training a model.\n", + "- Train a TensorFlow model.\n", + "- Retrieve and load the model artifacts.\n", + "- View the model evaluation.\n", + "- Upload the model as a Vertex `Model` resource.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for training and prediction.\n", + "\n", + "Set the variables `TRAIN_GPU/TRAIN_NGPU` and `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU.\n", + "\n", + "*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3 -- which is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n", + " TRAIN_GPU, TRAIN_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n", + " )\n", + "else:\n", + " TRAIN_GPU, TRAIN_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (None, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:training,prediction" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for training and prediction\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-3:latest`\n", + " - TensorFlow 2.4\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-4:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/training/xgboost-cpu.1-1`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/training/scikit-learn-cpu.0-23:latest`\n", + " - Pytorch\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-5:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-6:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-7:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for training](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers).\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-3:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-90:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-82:latest`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-23:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-22:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-20:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for prediction](https://cloud.google.com/vertex-ai/docs/predictions/pre-built-containers)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf2-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf2-cpu.{}\".format(TF)\n", + "else:\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "TRAIN_IMAGE = \"gcr.io/cloud-aiplatform/training/{}:latest\".format(TRAIN_VERSION)\n", + "DEPLOY_IMAGE = \"gcr.io/cloud-aiplatform/prediction/{}:latest\".format(DEPLOY_VERSION)\n", + "\n", + "print(\"Training:\", TRAIN_IMAGE, TRAIN_GPU, TRAIN_NGPU)\n", + "print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU, DEPLOY_NGPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:training,prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for training and prediction.\n", + "\n", + "- Set the variables `TRAIN_COMPUTE` and `DEPLOY_COMPUTE` to configure the compute resources for the VMs you will use for for training and prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: The following is not supported for training:*\n", + "\n", + " - `standard`: 2 vCPUs\n", + " - `highcpu`: 2, 4 and 8 vCPUs\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_TRAIN_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Train machine type\", TRAIN_COMPUTE)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own custom model and training for CIFAR10." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:custom" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Model Service for `Model` resources.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:custom" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"job\"] = create_job_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_model" + }, + "source": [ + "## Train a model\n", + "\n", + "There are two ways you can train a custom model using a container image:\n", + "\n", + "- **Use a Google Cloud prebuilt container**. If you use a prebuilt container, you will additionally specify a Python package to install into the container image. This Python package contains your code for training a custom model.\n", + "\n", + "- **Use your own custom container image**. If you use your own container, the container needs to contain your code for training a custom model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_specification:prebuilt_container" + }, + "source": [ + "## Prepare your custom job specification\n", + "\n", + "Now that your clients are ready, your first step is to create a Job Specification for your custom training job. The job specification will consist of the following:\n", + "\n", + "- `worker_pool_spec` : The specification of the type of machine(s) you will use for training and how many (single or distributed)\n", + "- `python_package_spec` : The specification of the Python package to be installed with the pre-built container." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "source": [ + "### Prepare your machine specification\n", + "\n", + "Now define the machine specification for your custom training job. This tells Vertex what type of machine instance to provision for the training.\n", + " - `machine_type`: The type of GCP instance to provision -- e.g., n1-standard-8.\n", + " - `accelerator_type`: The type, if any, of hardware accelerator. In this tutorial if you previously set the variable `TRAIN_GPU != None`, you are using a GPU; otherwise you will use a CPU.\n", + " - `accelerator_count`: The number of accelerators." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "outputs": [], + "source": [ + "if TRAIN_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": TRAIN_COMPUTE,\n", + " \"accelerator_type\": TRAIN_GPU,\n", + " \"accelerator_count\": TRAIN_NGPU,\n", + " }\n", + "else:\n", + " machine_spec = {\"machine_type\": TRAIN_COMPUTE, \"accelerator_count\": 0}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "source": [ + "### Prepare your disk specification\n", + "\n", + "(optional) Now define the disk specification for your custom training job. This tells Vertex what type and size of disk to provision in each machine instance for the training.\n", + "\n", + " - `boot_disk_type`: Either SSD or Standard. SSD is faster, and Standard is less expensive. Defaults to SSD.\n", + " - `boot_disk_size_gb`: Size of disk in GB." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "outputs": [], + "source": [ + "DISK_TYPE = \"pd-ssd\" # [ pd-ssd, pd-standard]\n", + "DISK_SIZE = 200 # GB\n", + "\n", + "disk_spec = {\"boot_disk_type\": DISK_TYPE, \"boot_disk_size_gb\": DISK_SIZE}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container" + }, + "source": [ + "### Define the worker pool specification\n", + "\n", + "Next, you define the worker pool specification for your custom training job. The worker pool specification will consist of the following:\n", + "\n", + "- `replica_count`: The number of instances to provision of this machine type.\n", + "- `machine_spec`: The hardware specification.\n", + "- `disk_spec` : (optional) The disk storage specification.\n", + "\n", + "- `python_package`: The Python training package to install on the VM instance(s) and which Python module to invoke, along with command line arguments for the Python module.\n", + "\n", + "Let's dive deeper now into the python package specification:\n", + "\n", + "-`executor_image_spec`: This is the docker image which is configured for your custom training job.\n", + "\n", + "-`package_uris`: This is a list of the locations (URIs) of your python training packages to install on the provisioned instance. The locations need to be in a Cloud Storage bucket. These can be either individual python files or a zip (archive) of an entire package. In the later case, the job service will unzip (unarchive) the contents into the docker image.\n", + "\n", + "-`python_module`: The Python module (script) to invoke for running the custom training job. In this example, you will be invoking `trainer.task.py` -- note that it was not neccessary to append the `.py` suffix.\n", + "\n", + "-`args`: The command line arguments to pass to the corresponding Pythom module. In this example, you will be setting:\n", + " - `\"--model-dir=\" + MODEL_DIR` : The Cloud Storage location where to store the model artifacts. There are two ways to tell the training script where to save the model artifacts:\n", + " - direct: You pass the Cloud Storage location as a command line argument to your training script (set variable `DIRECT = True`), or\n", + " - indirect: The service passes the Cloud Storage location as the environment variable `AIP_MODEL_DIR` to your training script (set variable `DIRECT = False`). In this case, you tell the service the model artifact location in the job specification.\n", + " - `\"--epochs=\" + EPOCHS`: The number of epochs for training.\n", + " - `\"--steps=\" + STEPS`: The number of steps (batches) per epoch.\n", + " - `\"--distribute=\" + TRAIN_STRATEGY\"` : The training distribution strategy to use for single or distributed training.\n", + " - `\"single\"`: single device.\n", + " - `\"mirror\"`: all GPU devices on a single compute instance.\n", + " - `\"multi\"`: all GPU devices on all compute instances." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_job_\" + TIMESTAMP\n", + "MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, JOB_NAME)\n", + "\n", + "if not TRAIN_NGPU or TRAIN_NGPU < 2:\n", + " TRAIN_STRATEGY = \"single\"\n", + "else:\n", + " TRAIN_STRATEGY = \"mirror\"\n", + "\n", + "EPOCHS = 20\n", + "STEPS = 100\n", + "\n", + "DIRECT = True\n", + "if DIRECT:\n", + " CMDARGS = [\n", + " \"--model-dir=\" + MODEL_DIR,\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "else:\n", + " CMDARGS = [\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "\n", + "worker_pool_spec = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": machine_spec,\n", + " \"disk_spec\": disk_spec,\n", + " \"python_package_spec\": {\n", + " \"executor_image_uri\": TRAIN_IMAGE,\n", + " \"package_uris\": [BUCKET_NAME + \"/trainer_cifar10.tar.gz\"],\n", + " \"python_module\": \"trainer.task\",\n", + " \"args\": CMDARGS,\n", + " },\n", + " }\n", + "]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "assemble_custom_job_specification" + }, + "source": [ + "### Assemble a job specification\n", + "\n", + "Now assemble the complete description for the custom job specification:\n", + "\n", + "- `display_name`: The human readable name you assign to this custom job.\n", + "- `job_spec`: The specification for the custom job.\n", + " - `worker_pool_specs`: The specification for the machine VM instances.\n", + " - `base_output_directory`: This tells the service the Cloud Storage location where to save the model artifacts (when variable `DIRECT = False`). The service will then pass the location to the training script as the environment variable `AIP_MODEL_DIR`, and the path will be of the form:\n", + "\n", + " /model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "assemble_custom_job_specification" + }, + "outputs": [], + "source": [ + "if DIRECT:\n", + " job_spec = {\"worker_pool_specs\": worker_pool_spec}\n", + "else:\n", + " job_spec = {\n", + " \"worker_pool_specs\": worker_pool_spec,\n", + " \"base_output_directory\": {\"output_uri_prefix\": MODEL_DIR},\n", + " }\n", + "\n", + "custom_job = {\"display_name\": JOB_NAME, \"job_spec\": job_spec}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "examine_training_package" + }, + "source": [ + "### Examine the training package\n", + "\n", + "#### Package layout\n", + "\n", + "Before you start the training, you will look at how a Python package is assembled for a custom training job. When unarchived, the package contains the following directory/file layout.\n", + "\n", + "- PKG-INFO\n", + "- README.md\n", + "- setup.cfg\n", + "- setup.py\n", + "- trainer\n", + " - \\_\\_init\\_\\_.py\n", + " - task.py\n", + "\n", + "The files `setup.cfg` and `setup.py` are the instructions for installing the package into the operating environment of the Docker image.\n", + "\n", + "The file `trainer/task.py` is the Python script for executing the custom training job. *Note*, when we referred to it in the worker pool specification, we replace the directory slash with a dot (`trainer.task`) and dropped the file suffix (`.py`).\n", + "\n", + "#### Package Assembly\n", + "\n", + "In the following cells, you will assemble the training package." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "examine_training_package" + }, + "outputs": [], + "source": [ + "# Make folder for Python training script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\ntag_build =\\n\\ntag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'tensorflow_datasets==1.3.0',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\nName: CIFAR10 image classification\\n\\nVersion: 0.0.0\\n\\nSummary: Demostration training script\\n\\nHome-page: www.google.com\\n\\nAuthor: Google\\n\\nAuthor-email: aferlitsch@google.com\\n\\nLicense: Public\\n\\nDescription: Demo\\n\\nPlatform: Vertex\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:cifar10" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the training script task.py. We won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n", + "- Loads CIFAR10 dataset from TF Datasets (tfds).\n", + "- Builds a model using TF.Keras model API.\n", + "- Compiles the model (`compile()`).\n", + "- Sets a training distribution strategy according to the argument `args.distribute`.\n", + "- Trains the model (`fit()`) with epochs and steps according to the arguments `args.epochs` and `args.steps`\n", + "- Saves the trained model (`save(args.model_dir)`) to the specified model directory." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:cifar10" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Single, Mirror and Multi-Machine Distributed Training for CIFAR-10\n", + "\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "import argparse\n", + "import os\n", + "import sys\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv(\"AIP_MODEL_DIR\"), type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.01, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=10, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=200, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "print('DEVICES', device_lib.list_local_devices())\n", + "\n", + "# Single Machine, single compute device\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "# Single Machine, multiple compute device\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "# Multiple Machine, multiple compute device\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "# Multi-worker configuration\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "# Preparing dataset\n", + "BUFFER_SIZE = 10000\n", + "BATCH_SIZE = 64\n", + "\n", + "\n", + "def make_datasets_unbatched():\n", + "\n", + " # Scaling CIFAR10 data from (0, 255] to (0., 1.]\n", + " def scale(image, label):\n", + " image = tf.cast(image, tf.float32)\n", + " image /= 255.0\n", + " return image, label\n", + "\n", + "\n", + " datasets, info = tfds.load(name='cifar10',\n", + " with_info=True,\n", + " as_supervised=True)\n", + " return datasets['train'].map(scale).cache().shuffle(BUFFER_SIZE).repeat()\n", + "\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_cnn_model():\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu', input_shape=(32, 32, 3)),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu'),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Flatten(),\n", + " tf.keras.layers.Dense(10, activation='softmax')\n", + " ])\n", + " model.compile(\n", + " loss=tf.keras.losses.sparse_categorical_crossentropy,\n", + " optimizer=tf.keras.optimizers.SGD(learning_rate=args.lr),\n", + " metrics=['accuracy'])\n", + " return model\n", + "\n", + "\n", + "# Train the model\n", + "NUM_WORKERS = strategy.num_replicas_in_sync\n", + "# Here the batch size scales up by number of workers since\n", + "# `tf.data.Dataset.batch` expects the global batch size.\n", + "GLOBAL_BATCH_SIZE = BATCH_SIZE * NUM_WORKERS\n", + "train_dataset = make_datasets_unbatched().batch(GLOBAL_BATCH_SIZE)\n", + "\n", + "with strategy.scope():\n", + " # Creation of dataset, and model building/compiling need to be within\n", + " # `strategy.scope()`.\n", + " model = build_and_compile_cnn_model()\n", + "\n", + "model.fit(x=train_dataset, epochs=args.epochs, steps_per_epoch=args.steps)\n", + "model.save(args.model_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tarball_training_script" + }, + "source": [ + "#### Store training script on your Cloud Storage bucket\n", + "\n", + "Next, you package the training folder into a compressed tar ball, and then store it in your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tarball_training_script" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_cifar10.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job" + }, + "source": [ + "### Train the model\n", + "\n", + "\n", + "Now start the training of your custom training job on Vertex. Use this helper function `create_custom_job`, which takes the following parameter:\n", + "\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "The helper function calls job client service's `create_custom_job` method, with the following parameters:\n", + "\n", + "-`parent`: The Vertex location path to `Dataset`, `Model` and `Endpoint` resources.\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "You will display a handful of the fields returned in `response` object, with the two that are of most interest are:\n", + "\n", + "`response.name`: The Vertex fully qualified identifier assigned to this custom training job. You save this identifier for using in subsequent steps.\n", + "\n", + "`response.state`: The current state of the custom training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job" + }, + "outputs": [], + "source": [ + "def create_custom_job(custom_job):\n", + " response = clients[\"job\"].create_custom_job(parent=PARENT, custom_job=custom_job)\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = create_custom_job(custom_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "job_id:response" + }, + "source": [ + "Now get the unique identifier for the custom job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom job\n", + "job_id = response.name\n", + "# The short numeric ID for the custom job\n", + "job_short_id = job_id.split(\"/\")[-1]\n", + "\n", + "print(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_custom_job" + }, + "source": [ + "### Get information on a custom job\n", + "\n", + "Next, use this helper function `get_custom_job`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "The helper function calls the job client service's`get_custom_job` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "If you recall, you got the Vertex fully qualified identifier for the custom job in the `response.name` field when you called the `create_custom_job` method, and saved the identifier in the variable `job_id`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_custom_job" + }, + "outputs": [], + "source": [ + "def get_custom_job(name, silent=False):\n", + " response = clients[\"job\"].get_custom_job(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = get_custom_job(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete:custom" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, we will need to know the location of the saved model, which the Python script saved in your local Cloud Storage bucket at `MODEL_DIR + '/saved_model.pb'`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete:custom" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_custom_job(job_id, True)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_path_to_deploy = None\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " if not DIRECT:\n", + " MODEL_DIR = MODEL_DIR + \"/model\"\n", + " model_path_to_deploy = MODEL_DIR\n", + " print(\"Training Time:\", response.update_time - response.create_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model_to_deploy:\", model_path_to_deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "load_saved_model" + }, + "source": [ + "## Load the saved model\n", + "\n", + "Your model is stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Now load it from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n", + "\n", + "To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "load_saved_model" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:image" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load the CIFAR10 test (holdout) data from `tf.keras.datasets`, using the method `load_data()`. This will return the dataset as a tuple of two elements. The first element is the training data and the second is the test data. Each element is also a tuple of two elements: the image data, and the corresponding labels.\n", + "\n", + "You don't need the training data, and hence why we loaded it as `(_, _)`.\n", + "\n", + "Before you can run the data through evaluation, you need to preprocess it:\n", + "\n", + "x_test:\n", + "1. Normalize (rescaling) the pixel data by dividing each pixel by 255. This will replace each single byte integer pixel with a 32-bit floating point number between 0 and 1.\n", + "\n", + "y_test:
\n", + "2. The labels are currently scalar (sparse). If you look back at the `compile()` step in the `trainer/task.py` script, you will find that it was compiled for sparse labels. So we don't need to do anything more." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "evaluate_custom_model:image,cifar10" + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "from tensorflow.keras.datasets import cifar10\n", + "\n", + "(_, _), (x_test, y_test) = cifar10.load_data()\n", + "x_test = (x_test / 255.0).astype(np.float32)\n", + "\n", + "print(x_test.shape, y_test.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "perform_evaluation_custom" + }, + "source": [ + "### Perform the model evaluation\n", + "\n", + "Now evaluate how well the model in the custom job did." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "perform_evaluation_custom" + }, + "outputs": [], + "source": [ + "model.evaluate(x_test, y_test)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "how_serving_function_works" + }, + "source": [ + "## Upload the model for serving\n", + "\n", + "Next, you will upload your TF.Keras model from the custom job to Vertex `Model` service, which will create a Vertex `Model` resource for your custom model. During upload, you need to define a serving function to convert data to the format your model expects. If you send encoded data to Vertex, your serving function ensures that the data is decoded on the model server before it is passed as input to your model.\n", + "\n", + "### How does the serving function work\n", + "\n", + "When you send a request to an online prediction server, the request is received by a HTTP server. The HTTP server extracts the prediction request from the HTTP request content body. The extracted prediction request is forwarded to the serving function. For Google pre-built prediction containers, the request content is passed to the serving function as a `tf.string`.\n", + "\n", + "The serving function consists of two parts:\n", + "\n", + "- `preprocessing function`:\n", + " - Converts the input (`tf.string`) to the input shape and data type of the underlying model (dynamic graph).\n", + " - Performs the same preprocessing of the data that was done during training the underlying model -- e.g., normalizing, scaling, etc.\n", + "- `post-processing function`:\n", + " - Converts the model output to format expected by the receiving application -- e.q., compresses the output.\n", + " - Packages the output for the the receiving application -- e.g., add headings, make JSON object, etc.\n", + "\n", + "Both the preprocessing and post-processing functions are converted to static graphs which are fused to the model. The output from the underlying model is passed to the post-processing function. The post-processing function passes the converted/packaged output back to the HTTP server. The HTTP server returns the output as the HTTP response content.\n", + "\n", + "One consideration you need to consider when building serving functions for TF.Keras models is that they run as static graphs. That means, you cannot use TF graph operations that require a dynamic graph. If you do, you will get an error during the compile of the serving function which will indicate that you are using an EagerTensor which is not supported." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_image" + }, + "source": [ + "### Serving function for image data\n", + "\n", + "To pass images to the prediction service, you encode the compressed (e.g., JPEG) image bytes into base 64 -- which makes the content safe from modification while transmitting binary data over the network. Since this deployed model expects input data as raw (uncompressed) bytes, you need to ensure that the base 64 encoded data gets converted back to raw bytes before it is passed as input to the deployed model.\n", + "\n", + "To resolve this, define a serving function (`serving_fn`) and attach it to the model as a preprocessing step. Add a `@tf.function` decorator so the serving function is fused to the underlying model (instead of upstream on a CPU).\n", + "\n", + "When you send a prediction or explanation request, the content of the request is base 64 decoded into a Tensorflow string (`tf.string`), which is passed to the serving function (`serving_fn`). The serving function preprocesses the `tf.string` into raw (uncompressed) numpy bytes (`preprocess_fn`) to match the input requirements of the model:\n", + "- `io.decode_jpeg`- Decompresses the JPG image which is returned as a Tensorflow tensor with three channels (RGB).\n", + "- `image.convert_image_dtype` - Changes integer pixel values to float 32.\n", + "- `image.resize` - Resizes the image to match the input shape for the model.\n", + "- `resized / 255.0` - Rescales (normalization) the pixel data between 0 and 1.\n", + "\n", + "At this point, the data can be passed to the model (`m_call`)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_image" + }, + "outputs": [], + "source": [ + "CONCRETE_INPUT = \"numpy_inputs\"\n", + "\n", + "\n", + "def _preprocess(bytes_input):\n", + " decoded = tf.io.decode_jpeg(bytes_input, channels=3)\n", + " decoded = tf.image.convert_image_dtype(decoded, tf.float32)\n", + " resized = tf.image.resize(decoded, size=(32, 32))\n", + " rescale = tf.cast(resized / 255.0, tf.float32)\n", + " return rescale\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def preprocess_fn(bytes_inputs):\n", + " decoded_images = tf.map_fn(\n", + " _preprocess, bytes_inputs, dtype=tf.float32, back_prop=False\n", + " )\n", + " return {\n", + " CONCRETE_INPUT: decoded_images\n", + " } # User needs to make sure the key matches model's input\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def serving_fn(bytes_inputs):\n", + " images = preprocess_fn(bytes_inputs)\n", + " prob = m_call(**images)\n", + " return prob\n", + "\n", + "\n", + "m_call = tf.function(model.call).get_concrete_function(\n", + " [tf.TensorSpec(shape=[None, 32, 32, 3], dtype=tf.float32, name=CONCRETE_INPUT)]\n", + ")\n", + "\n", + "tf.saved_model.save(\n", + " model, model_path_to_deploy, signatures={\"serving_default\": serving_fn}\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_signature:image" + }, + "source": [ + "## Get the serving function signature\n", + "\n", + "You can get the signatures of your model's input and output layers by reloading the model into memory, and querying it for the signatures corresponding to each layer.\n", + "\n", + "For your purpose, you need the signature of the serving function. Why? Well, when we send our data for prediction as a HTTP request packet, the image data is base64 encoded, and our TF.Keras model takes numpy input. Your serving function will do the conversion from base64 to a numpy array.\n", + "\n", + "When making a prediction request, you need to route the request to the serving function instead of the model, so you need to know the input layer name of the serving function -- which you will use later when you make a prediction request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_signature:image" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_path_to_deploy)\n", + "\n", + "serving_input = list(\n", + " loaded.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "print(\"Serving function input:\", serving_input)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "upload_the_model" + }, + "source": [ + "### Upload the model\n", + "\n", + "Use this helper function `upload_model` to upload your model, stored in SavedModel format, up to the `Model` service, which will instantiate a Vertex `Model` resource instance for your model. Once you've done that, you can use the `Model` resource instance in the same way as any other Vertex `Model` resource instance, such as deploying to an `Endpoint` resource for serving predictions.\n", + "\n", + "The helper function takes the following parameters:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` service.\n", + "- `image_uri`: The container image for the model deployment.\n", + "- `model_uri`: The Cloud Storage path to our SavedModel artificat. For this tutorial, this is the Cloud Storage location where the `trainer/task.py` saved the model artifacts, which we specified in the variable `MODEL_DIR`.\n", + "\n", + "The helper function calls the `Model` client service's method `upload_model`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for `Dataset`, `Model` and `Endpoint` resources.\n", + "- `model`: The specification for the Vertex `Model` resource instance.\n", + "\n", + "Let's now dive deeper into the Vertex model specification `model`. This is a dictionary object that consists of the following fields:\n", + "\n", + "- `display_name`: A human readable name for the `Model` resource.\n", + "- `metadata_schema_uri`: Since your model was built without an Vertex `Dataset` resource, you will leave this blank (`''`).\n", + "- `artificat_uri`: The Cloud Storage path where the model is stored in SavedModel format.\n", + "- `container_spec`: This is the specification for the Docker container that will be installed on the `Endpoint` resource, from which the `Model` resource will serve predictions. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + "\n", + "Uploading a model into a Vertex Model resource returns a long running operation, since it may take a few moments. You call response.result(), which is a synchronous call and will return when the Vertex Model resource is ready.\n", + "\n", + "The helper function returns the Vertex fully qualified identifier for the corresponding Vertex Model instance upload_model_response.model. You will save the identifier for subsequent steps in the variable model_to_deploy_id." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "upload_the_model" + }, + "outputs": [], + "source": [ + "IMAGE_URI = DEPLOY_IMAGE\n", + "\n", + "\n", + "def upload_model(display_name, image_uri, model_uri):\n", + " model = {\n", + " \"display_name\": display_name,\n", + " \"metadata_schema_uri\": \"\",\n", + " \"artifact_uri\": model_uri,\n", + " \"container_spec\": {\n", + " \"image_uri\": image_uri,\n", + " \"command\": [],\n", + " \"args\": [],\n", + " \"env\": [{\"name\": \"env_name\", \"value\": \"env_value\"}],\n", + " \"ports\": [{\"container_port\": 8080}],\n", + " \"predict_route\": \"\",\n", + " \"health_route\": \"\",\n", + " },\n", + " }\n", + " response = clients[\"model\"].upload_model(parent=PARENT, model=model)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " upload_model_response = response.result(timeout=180)\n", + " print(\"upload_model_response\")\n", + " print(\" model:\", upload_model_response.model)\n", + " return upload_model_response.model\n", + "\n", + "\n", + "model_to_deploy_id = upload_model(\n", + " \"cifar10-\" + TIMESTAMP, IMAGE_URI, model_path_to_deploy\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_model" + }, + "source": [ + "### Get `Model` resource information\n", + "\n", + "Now let's get the model information for just your model. Use this helper function `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource.\n", + "\n", + "This helper function calls the Vertex `Model` client service's method `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_model" + }, + "outputs": [], + "source": [ + "def get_model(name):\n", + " response = clients[\"model\"].get_model(name=name)\n", + " print(response)\n", + "\n", + "\n", + "get_model(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:custom" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex custom `Model` resource. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"cifar10_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:dedicated" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `dedicated_resources`: This refers to how many compute instances (replicas) that are scaled for serving prediction requests.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `min_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:dedicated" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"cifar10_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " \"machine_spec\": machine_spec,\n", + " },\n", + " \"disable_container_logging\": False,\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_item:test" + }, + "source": [ + "### Get test item\n", + "\n", + "You will use an example out of the test (holdout) portion of the dataset as a test item." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:test" + }, + "outputs": [], + "source": [ + "test_image = x_test[0]\n", + "test_label = y_test[0]\n", + "print(test_image.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "prepare_test_item:test,image" + }, + "source": [ + "### Prepare the request content\n", + "You are going to send the CIFAR10 image as compressed JPG image, instead of the raw uncompressed bytes:\n", + "\n", + "- `cv2.imwrite`: Use openCV to write the uncompressed image to disk as a compressed JPEG image.\n", + " - Denormalize the image data from \\[0,1) range back to [0,255).\n", + " - Convert the 32-bit floating point values to 8-bit unsigned integers.\n", + "- `tf.io.read_file`: Read the compressed JPG images back into memory as raw bytes.\n", + "- `base64.b64encode`: Encode the raw bytes into a base 64 encoded string." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "prepare_test_item:test,image" + }, + "outputs": [], + "source": [ + "import base64\n", + "\n", + "import cv2\n", + "\n", + "cv2.imwrite(\"tmp.jpg\", (test_image * 255).astype(np.uint8))\n", + "\n", + "bytes = tf.io.read_file(\"tmp.jpg\")\n", + "b64str = base64.b64encode(bytes.numpy()).decode(\"utf-8\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "send_prediction_request:image" + }, + "source": [ + "### Send the prediction request\n", + "\n", + "Ok, now you have a test image. Use this helper function `predict_image`, which takes the following parameters:\n", + "\n", + "- `image`: The test image data as a numpy array.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed to.\n", + "- `parameters_dict`: Additional parameters for serving.\n", + "\n", + "This function calls the prediction client service `predict` method with the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed to.\n", + "- `instances`: A list of instances (encoded images) to predict.\n", + "- `parameters`: Additional parameters for serving.\n", + "\n", + "To pass the image data to the prediction service, in the previous step you encoded the bytes into base64 -- which makes the content safe from modification when transmitting binary data over the network. You need to tell the serving binary where your model is deployed to, that the content has been base64 encoded, so it will decode it on the other end in the serving binary.\n", + "\n", + "Each instance in the prediction request is a dictionary entry of the form:\n", + "\n", + " {serving_input: {'b64': content}}\n", + "\n", + "- `input_name`: the name of the input layer of the underlying model.\n", + "- `'b64'`: A key that indicates the content is base64 encoded.\n", + "- `content`: The compressed JPG image bytes as a base64 encoded string.\n", + "\n", + "Since the `predict()` service can take multiple images (instances), you will send your single image as a list of one image. As a final step, you package the instances list into Google's protobuf format -- which is what we pass to the `predict()` service.\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding image in the request. You will see in the output for each prediction:\n", + "\n", + "- `predictions`: Confidence level for the prediction, between 0 and 1, for each of the classes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "send_prediction_request:image" + }, + "outputs": [], + "source": [ + "def predict_image(image, endpoint, parameters_dict):\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [{serving_input: {\"b64\": image}}]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters_dict\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", prediction)\n", + "\n", + "\n", + "predict_image(b64str, endpoint_id, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_custom_image_classification_online.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_custom_image_classification_online_ab_testing.ipynb b/notebooks/community/gapic/custom/showcase_custom_image_classification_online_ab_testing.ipynb new file mode 100644 index 000000000..7015e9fe0 --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_custom_image_classification_online_ab_testing.ipynb @@ -0,0 +1,2444 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Custom training image classification model for online prediction for A/B testing\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom,ab" + }, + "source": [ + "## Overview\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to train and deploy a custom image classification model for A/B testing for online prediction." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:custom,cifar10,icn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [CIFAR10 dataset](https://www.tensorflow.org/datasets/catalog/cifar10) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use is built into TensorFlow. The trained model predicts which type of class an image is from ten classes: airplane, automobile, bird, cat, deer, dog, frog, horse, ship, truck." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,ab,online_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you learn how to create multiple instances of a custom model from a Python script in a Docker container using the Vertex client library and then deploy for A/B testing\n", + "of online predictions. You can alternatively create custom models from the command line using gcloud or online using Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create an Vertex custom job for training a model.\n", + "- Train two instances (A and B) of the TensorFlow model.\n", + "- Retrieve and load the models artifacts.\n", + "- View the models evaluation.\n", + "- Upload each model instance as a Vertex `Model` resource.\n", + "- Deploy the model instances to the same serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Review the results from the two model instances.\n", + "- Undeploy the `Model` resources." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for training and prediction.\n", + "\n", + "Set the variables `TRAIN_GPU/TRAIN_NGPU` and `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU.\n", + "\n", + "*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3 -- which is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n", + " TRAIN_GPU, TRAIN_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n", + " )\n", + "else:\n", + " TRAIN_GPU, TRAIN_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (None, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:training,prediction" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for training and prediction\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-3:latest`\n", + " - TensorFlow 2.4\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-4:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/training/xgboost-cpu.1-1`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/training/scikit-learn-cpu.0-23:latest`\n", + " - Pytorch\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-5:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-6:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-7:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for training](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers).\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-3:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-90:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-82:latest`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-23:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-22:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-20:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for prediction](https://cloud.google.com/vertex-ai/docs/predictions/pre-built-containers)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf2-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf2-cpu.{}\".format(TF)\n", + "else:\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "TRAIN_IMAGE = \"gcr.io/cloud-aiplatform/training/{}:latest\".format(TRAIN_VERSION)\n", + "DEPLOY_IMAGE = \"gcr.io/cloud-aiplatform/prediction/{}:latest\".format(DEPLOY_VERSION)\n", + "\n", + "print(\"Training:\", TRAIN_IMAGE, TRAIN_GPU, TRAIN_NGPU)\n", + "print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU, DEPLOY_NGPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:training,prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for training and prediction.\n", + "\n", + "- Set the variables `TRAIN_COMPUTE` and `DEPLOY_COMPUTE` to configure the compute resources for the VMs you will use for for training and prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: The following is not supported for training:*\n", + "\n", + " - `standard`: 2 vCPUs\n", + " - `highcpu`: 2, 4 and 8 vCPUs\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_TRAIN_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Train machine type\", TRAIN_COMPUTE)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own custom model and training for CIFAR10." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:custom" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Model Service for `Model` resources.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:custom" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"job\"] = create_job_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_model" + }, + "source": [ + "## Train a model\n", + "\n", + "There are two ways you can train a custom model using a container image:\n", + "\n", + "- **Use a Google Cloud prebuilt container**. If you use a prebuilt container, you will additionally specify a Python package to install into the container image. This Python package contains your code for training a custom model.\n", + "\n", + "- **Use your own custom container image**. If you use your own container, the container needs to contain your code for training a custom model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_specification:prebuilt_container" + }, + "source": [ + "## Prepare your custom job specification\n", + "\n", + "Now that your clients are ready, your first step is to create a Job Specification for your custom training job. The job specification will consist of the following:\n", + "\n", + "- `worker_pool_spec` : The specification of the type of machine(s) you will use for training and how many (single or distributed)\n", + "- `python_package_spec` : The specification of the Python package to be installed with the pre-built container." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "source": [ + "### Prepare your machine specification\n", + "\n", + "Now define the machine specification for your custom training job. This tells Vertex what type of machine instance to provision for the training.\n", + " - `machine_type`: The type of GCP instance to provision -- e.g., n1-standard-8.\n", + " - `accelerator_type`: The type, if any, of hardware accelerator. In this tutorial if you previously set the variable `TRAIN_GPU != None`, you are using a GPU; otherwise you will use a CPU.\n", + " - `accelerator_count`: The number of accelerators." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "outputs": [], + "source": [ + "if TRAIN_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": TRAIN_COMPUTE,\n", + " \"accelerator_type\": TRAIN_GPU,\n", + " \"accelerator_count\": TRAIN_NGPU,\n", + " }\n", + "else:\n", + " machine_spec = {\"machine_type\": TRAIN_COMPUTE, \"accelerator_count\": 0}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "source": [ + "### Prepare your disk specification\n", + "\n", + "(optional) Now define the disk specification for your custom training job. This tells Vertex what type and size of disk to provision in each machine instance for the training.\n", + "\n", + " - `boot_disk_type`: Either SSD or Standard. SSD is faster, and Standard is less expensive. Defaults to SSD.\n", + " - `boot_disk_size_gb`: Size of disk in GB." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "outputs": [], + "source": [ + "DISK_TYPE = \"pd-ssd\" # [ pd-ssd, pd-standard]\n", + "DISK_SIZE = 200 # GB\n", + "\n", + "disk_spec = {\"boot_disk_type\": DISK_TYPE, \"boot_disk_size_gb\": DISK_SIZE}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "examine_training_package" + }, + "source": [ + "### Examine the training package\n", + "\n", + "#### Package layout\n", + "\n", + "Before you start the training, you will look at how a Python package is assembled for a custom training job. When unarchived, the package contains the following directory/file layout.\n", + "\n", + "- PKG-INFO\n", + "- README.md\n", + "- setup.cfg\n", + "- setup.py\n", + "- trainer\n", + " - \\_\\_init\\_\\_.py\n", + " - task.py\n", + "\n", + "The files `setup.cfg` and `setup.py` are the instructions for installing the package into the operating environment of the Docker image.\n", + "\n", + "The file `trainer/task.py` is the Python script for executing the custom training job. *Note*, when we referred to it in the worker pool specification, we replace the directory slash with a dot (`trainer.task`) and dropped the file suffix (`.py`).\n", + "\n", + "#### Package Assembly\n", + "\n", + "In the following cells, you will assemble the training package." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "examine_training_package" + }, + "outputs": [], + "source": [ + "# Make folder for Python training script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\ntag_build =\\n\\ntag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'tensorflow_datasets==1.3.0',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\nName: CIFAR10 image classification\\n\\nVersion: 0.0.0\\n\\nSummary: Demostration training script\\n\\nHome-page: www.google.com\\n\\nAuthor: Google\\n\\nAuthor-email: aferlitsch@google.com\\n\\nLicense: Public\\n\\nDescription: Demo\\n\\nPlatform: Vertex\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:cifar10" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the training script task.py. We won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n", + "- Loads CIFAR10 dataset from TF Datasets (tfds).\n", + "- Builds a model using TF.Keras model API.\n", + "- Compiles the model (`compile()`).\n", + "- Sets a training distribution strategy according to the argument `args.distribute`.\n", + "- Trains the model (`fit()`) with epochs and steps according to the arguments `args.epochs` and `args.steps`\n", + "- Saves the trained model (`save(args.model_dir)`) to the specified model directory." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:cifar10" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Single, Mirror and Multi-Machine Distributed Training for CIFAR-10\n", + "\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "import argparse\n", + "import os\n", + "import sys\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv(\"AIP_MODEL_DIR\"), type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.01, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=10, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=200, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "print('DEVICES', device_lib.list_local_devices())\n", + "\n", + "# Single Machine, single compute device\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "# Single Machine, multiple compute device\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "# Multiple Machine, multiple compute device\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "# Multi-worker configuration\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "# Preparing dataset\n", + "BUFFER_SIZE = 10000\n", + "BATCH_SIZE = 64\n", + "\n", + "\n", + "def make_datasets_unbatched():\n", + "\n", + " # Scaling CIFAR10 data from (0, 255] to (0., 1.]\n", + " def scale(image, label):\n", + " image = tf.cast(image, tf.float32)\n", + " image /= 255.0\n", + " return image, label\n", + "\n", + "\n", + " datasets, info = tfds.load(name='cifar10',\n", + " with_info=True,\n", + " as_supervised=True)\n", + " return datasets['train'].map(scale).cache().shuffle(BUFFER_SIZE).repeat()\n", + "\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_cnn_model():\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu', input_shape=(32, 32, 3)),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu'),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Flatten(),\n", + " tf.keras.layers.Dense(10, activation='softmax')\n", + " ])\n", + " model.compile(\n", + " loss=tf.keras.losses.sparse_categorical_crossentropy,\n", + " optimizer=tf.keras.optimizers.SGD(learning_rate=args.lr),\n", + " metrics=['accuracy'])\n", + " return model\n", + "\n", + "\n", + "# Train the model\n", + "NUM_WORKERS = strategy.num_replicas_in_sync\n", + "# Here the batch size scales up by number of workers since\n", + "# `tf.data.Dataset.batch` expects the global batch size.\n", + "GLOBAL_BATCH_SIZE = BATCH_SIZE * NUM_WORKERS\n", + "train_dataset = make_datasets_unbatched().batch(GLOBAL_BATCH_SIZE)\n", + "\n", + "with strategy.scope():\n", + " # Creation of dataset, and model building/compiling need to be within\n", + " # `strategy.scope()`.\n", + " model = build_and_compile_cnn_model()\n", + "\n", + "model.fit(x=train_dataset, epochs=args.epochs, steps_per_epoch=args.steps)\n", + "model.save(args.model_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tarball_training_script" + }, + "source": [ + "#### Store training script on your Cloud Storage bucket\n", + "\n", + "Next, you package the training folder into a compressed tar ball, and then store it in your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tarball_training_script" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_cifar10.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container,a" + }, + "source": [ + "### Define the worker pool specification for Model A\n", + "\n", + "Next, you define the worker pool specification for your custom training job. The worker pool specification will consist of the following:\n", + "\n", + "- `replica_count`: The number of instances to provision of this machine type.\n", + "- `machine_spec`: The hardware specification.\n", + "- `disk_spec` : (optional) The disk storage specification.\n", + "\n", + "- `python_package`: The Python training package to install on the VM instance(s) and which Python module to invoke, along with command line arguments for the Python module.\n", + "\n", + "Let's dive deeper now into the python package specification:\n", + "\n", + "-`executor_image_spec`: This is the docker image which is configured for your custom training job.\n", + "\n", + "-`package_uris`: This is a list of the locations (URIs) of your python training packages to install on the provisioned instance. The locations need to be in a Cloud Storage bucket. These can be either individual python files or a zip (archive) of an entire package. In the later case, the job service will unzip (unarchive) the contents into the docker image.\n", + "\n", + "-`python_module`: The Python module (script) to invoke for running the custom training job. In this example, you will be invoking `trainer.task.py` -- note that it was not neccessary to append the `.py` suffix.\n", + "\n", + "-`args`: The command line arguments to pass to the corresponding Pythom module. In this example, you will be setting:\n", + " - `\"--model-dir=\" + MODEL_DIR` : The Cloud Storage location where to store the model artifacts. There are two ways to tell the training script where to save the model artifacts:\n", + " - direct: You pass the Cloud Storage location as a command line argument to your training script (set variable `DIRECT = True`), or\n", + " - indirect: The service passes the Cloud Storage location as the environment variable `AIP_MODEL_DIR` to your training script (set variable `DIRECT = False`). In this case, you tell the service the model artifact location in the job specification.\n", + " - `\"--epochs=\" + EPOCHS`: The number of epochs for training.\n", + " - `\"--steps=\" + STEPS`: The number of steps (batches) per epoch.\n", + " - `\"--distribute=\" + TRAIN_STRATEGY\"` : The training distribution strategy to use for single or distributed training.\n", + " - `\"single\"`: single device.\n", + " - `\"mirror\"`: all GPU devices on a single compute instance.\n", + " - `\"multi\"`: all GPU devices on all compute instances." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container,a" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_job_A\" + TIMESTAMP\n", + "MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, JOB_NAME)\n", + "MODEL_DIR_A = MODEL_DIR\n", + "\n", + "if not TRAIN_NGPU or TRAIN_NGPU < 2:\n", + " TRAIN_STRATEGY = \"single\"\n", + "else:\n", + " TRAIN_STRATEGY = \"mirror\"\n", + "\n", + "EPOCHS = 20\n", + "STEPS = 100\n", + "\n", + "DIRECT = True\n", + "if DIRECT:\n", + " CMDARGS = [\n", + " \"--model-dir=\" + MODEL_DIR,\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "else:\n", + " CMDARGS = [\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "\n", + "worker_pool_spec = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": machine_spec,\n", + " \"disk_spec\": disk_spec,\n", + " \"python_package_spec\": {\n", + " \"executor_image_uri\": TRAIN_IMAGE,\n", + " \"package_uris\": [BUCKET_NAME + \"/trainer_cifar10.tar.gz\"],\n", + " \"python_module\": \"trainer.task\",\n", + " \"args\": CMDARGS,\n", + " },\n", + " }\n", + "]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "assemble_custom_job_specification" + }, + "source": [ + "### Assemble a job specification\n", + "\n", + "Now assemble the complete description for the custom job specification:\n", + "\n", + "- `display_name`: The human readable name you assign to this custom job.\n", + "- `job_spec`: The specification for the custom job.\n", + " - `worker_pool_specs`: The specification for the machine VM instances.\n", + " - `base_output_directory`: This tells the service the Cloud Storage location where to save the model artifacts (when variable `DIRECT = False`). The service will then pass the location to the training script as the environment variable `AIP_MODEL_DIR`, and the path will be of the form:\n", + "\n", + " /model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "assemble_custom_job_specification" + }, + "outputs": [], + "source": [ + "if DIRECT:\n", + " job_spec = {\"worker_pool_specs\": worker_pool_spec}\n", + "else:\n", + " job_spec = {\n", + " \"worker_pool_specs\": worker_pool_spec,\n", + " \"base_output_directory\": {\"output_uri_prefix\": MODEL_DIR},\n", + " }\n", + "\n", + "custom_job = {\"display_name\": JOB_NAME, \"job_spec\": job_spec}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job" + }, + "source": [ + "### Train the model\n", + "\n", + "\n", + "Now start the training of your custom training job on Vertex. Use this helper function `create_custom_job`, which takes the following parameter:\n", + "\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "The helper function calls job client service's `create_custom_job` method, with the following parameters:\n", + "\n", + "-`parent`: The Vertex location path to `Dataset`, `Model` and `Endpoint` resources.\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "You will display a handful of the fields returned in `response` object, with the two that are of most interest are:\n", + "\n", + "`response.name`: The Vertex fully qualified identifier assigned to this custom training job. You save this identifier for using in subsequent steps.\n", + "\n", + "`response.state`: The current state of the custom training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job" + }, + "outputs": [], + "source": [ + "def create_custom_job(custom_job):\n", + " response = clients[\"job\"].create_custom_job(parent=PARENT, custom_job=custom_job)\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = create_custom_job(custom_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "job_id:response" + }, + "source": [ + "Now get the unique identifier for the custom job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom job\n", + "job_id = response.name\n", + "# The short numeric ID for the custom job\n", + "job_short_id = job_id.split(\"/\")[-1]\n", + "\n", + "print(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_custom_job" + }, + "source": [ + "### Get information on a custom job\n", + "\n", + "Next, use this helper function `get_custom_job`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "The helper function calls the job client service's`get_custom_job` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "If you recall, you got the Vertex fully qualified identifier for the custom job in the `response.name` field when you called the `create_custom_job` method, and saved the identifier in the variable `job_id`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_custom_job" + }, + "outputs": [], + "source": [ + "def get_custom_job(name, silent=False):\n", + " response = clients[\"job\"].get_custom_job(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = get_custom_job(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete:custom,a" + }, + "source": [ + "## Wait for training to complete\n", + "\n", + "Training the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, we will need to know the location of the saved model, which the Python script saved in your local Cloud Storage bucket at `MODEL_DIR + '/saved_model.pb'`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete:custom,a" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_custom_job(job_id, True)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_path_to_deploy_A = None\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " if not DIRECT:\n", + " MODEL_DIR_A = MODEL_DIR_A + \"/model\"\n", + " model_path_to_deploy_A = MODEL_DIR_A\n", + " print(\"Training Time:\", response.update_time - response.create_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model_to_deploy:\", model_path_to_deploy_A)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container,b" + }, + "source": [ + "### Define the worker pool specification for Model B\n", + "\n", + "Next, you define the worker pool specification for your custom training job. The worker pool specification will consist of the following:\n", + "\n", + "- `replica_count`: The number of instances to provision of this machine type.\n", + "- `machine_spec`: The hardware specification.\n", + "- `disk_spec` : (optional) The disk storage specification." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container,b" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_job_B\" + TIMESTAMP\n", + "MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, JOB_NAME)\n", + "MODEL_DIR_B = MODEL_DIR\n", + "\n", + "if not TRAIN_NGPU or TRAIN_NGPU < 2:\n", + " TRAIN_STRATEGY = \"single\"\n", + "else:\n", + " TRAIN_STRATEGY = \"mirror\"\n", + "\n", + "EPOCHS = 20\n", + "STEPS = 100\n", + "\n", + "DIRECT = True\n", + "if DIRECT:\n", + " CMDARGS = [\n", + " \"--model-dir=\" + MODEL_DIR,\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "else:\n", + " CMDARGS = [\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "\n", + "worker_pool_spec = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": machine_spec,\n", + " \"disk_spec\": disk_spec,\n", + " \"python_package_spec\": {\n", + " \"executor_image_uri\": TRAIN_IMAGE,\n", + " \"package_uris\": [BUCKET_NAME + \"/trainer_cifar10.tar.gz\"],\n", + " \"python_module\": \"trainer.task\",\n", + " \"args\": CMDARGS,\n", + " },\n", + " }\n", + "]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "assemble_custom_job_specification" + }, + "source": [ + "### Assemble a job specification\n", + "\n", + "Now assemble the complete description for the custom job specification:\n", + "\n", + "- `display_name`: The human readable name you assign to this custom job.\n", + "- `job_spec`: The specification for the custom job.\n", + " - `worker_pool_specs`: The specification for the machine VM instances.\n", + " - `base_output_directory`: This tells the service the Cloud Storage location where to save the model artifacts (when variable `DIRECT = False`). The service will then pass the location to the training script as the environment variable `AIP_MODEL_DIR`, and the path will be of the form:\n", + "\n", + " /model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "assemble_custom_job_specification" + }, + "outputs": [], + "source": [ + "if DIRECT:\n", + " job_spec = {\"worker_pool_specs\": worker_pool_spec}\n", + "else:\n", + " job_spec = {\n", + " \"worker_pool_specs\": worker_pool_spec,\n", + " \"base_output_directory\": {\"output_uri_prefix\": MODEL_DIR},\n", + " }\n", + "\n", + "custom_job = {\"display_name\": JOB_NAME, \"job_spec\": job_spec}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job" + }, + "source": [ + "### Train the model\n", + "\n", + "\n", + "Now start the training of your custom training job on Vertex. Use this helper function `create_custom_job`, which takes the following parameter:\n", + "\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "The helper function calls job client service's `create_custom_job` method, with the following parameters:\n", + "\n", + "-`parent`: The Vertex location path to `Dataset`, `Model` and `Endpoint` resources.\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "You will display a handful of the fields returned in `response` object, with the two that are of most interest are:\n", + "\n", + "`response.name`: The Vertex fully qualified identifier assigned to this custom training job. You save this identifier for using in subsequent steps.\n", + "\n", + "`response.state`: The current state of the custom training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job" + }, + "outputs": [], + "source": [ + "def create_custom_job(custom_job):\n", + " response = clients[\"job\"].create_custom_job(parent=PARENT, custom_job=custom_job)\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = create_custom_job(custom_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "job_id:response" + }, + "source": [ + "Now get the unique identifier for the custom job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom job\n", + "job_id = response.name\n", + "# The short numeric ID for the custom job\n", + "job_short_id = job_id.split(\"/\")[-1]\n", + "\n", + "print(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete:custom,b" + }, + "source": [ + "## Wait for training to complete\n", + "\n", + "Training the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, we will need to know the location of the saved model, which the Python script saved in your local Cloud Storage bucket at `MODEL_DIR + '/saved_model.pb'`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete:custom,b" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_custom_job(job_id, True)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_path_to_deploy_B = None\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " if not DIRECT:\n", + " MODEL_DIR_B = MODEL_DIR_B + \"/model\"\n", + " model_path_to_deploy_B = MODEL_DIR_A\n", + " print(\"Training Time:\", response.update_time - response.create_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model_to_deploy:\", model_path_to_deploy_B)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "load_saved_model:ab_testing" + }, + "source": [ + "## Load the saved model\n", + "\n", + "Your model instances are stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Let's go ahead and load them from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n", + "\n", + "To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR_A` and `MODEL_DIR_B`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "load_saved_model:ab_testing" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model_A = tf.keras.models.load_model(MODEL_DIR_A)\n", + "model_B = tf.keras.models.load_model(MODEL_DIR_B)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:image" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load the CIFAR10 test (holdout) data from `tf.keras.datasets`, using the method `load_data()`. This will return the dataset as a tuple of two elements. The first element is the training data and the second is the test data. Each element is also a tuple of two elements: the image data, and the corresponding labels.\n", + "\n", + "You don't need the training data, and hence why we loaded it as `(_, _)`.\n", + "\n", + "Before you can run the data through evaluation, you need to preprocess it:\n", + "\n", + "x_test:\n", + "1. Normalize (rescaling) the pixel data by dividing each pixel by 255. This will replace each single byte integer pixel with a 32-bit floating point number between 0 and 1.\n", + "\n", + "y_test:
\n", + "2. The labels are currently scalar (sparse). If you look back at the `compile()` step in the `trainer/task.py` script, you will find that it was compiled for sparse labels. So we don't need to do anything more." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "evaluate_custom_model:image,cifar10" + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "from tensorflow.keras.datasets import cifar10\n", + "\n", + "(_, _), (x_test, y_test) = cifar10.load_data()\n", + "x_test = (x_test / 255.0).astype(np.float32)\n", + "\n", + "print(x_test.shape, y_test.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ab_testing_wrapper" + }, + "source": [ + "## Adding client instance to model outputs.\n", + "\n", + "For A/B testing, each model needs to output two items in addition to the prediction:\n", + "\n", + "- The model instance, whether it is A or B.\n", + "- An identifier for the client session where the prediction request originates from.\n", + "\n", + "The model identifier is already baked into the prediction result returned by the `predict()` method, and you can use the model identifier as the means to determine which model instance A or B made the prediction.\n", + "\n", + "Now, why do you need to know the client session? In A/B testing, your not comparing the model's objective performance -- you've done that already in both model evaluation and post in continuous evaluation. Your comparing a business objective, such as did the customer click through the display ad, did they select a recommendation, was there a transaction conversion, etc. Thus, the business objective is measured on the client session, and you have to associate the model instance with the client session.\n", + "\n", + "\n", + "### Adding client session output for A/B Testing\n", + "\n", + "In the TF.Keras Functional API, when we build the model using the Model() class, we pass two parameters, the input tensor and the output layer; which I call pulling it all together connecting the inputs to the outputs:\n", + "\n", + "```\n", + "my_model = Model(inputs, outputs)\n", + "```\n", + "\n", + "We will use this method to implement passing through client session identification at prediction with your trained model instances. The syntax for specifying both multiple inputs and outputs looks like this:\n", + "\n", + "```\n", + "my_model = Model( inputs, [outputs1, outputs2])\n", + "```\n", + "\n", + "This assumes that the application server, which makes the prediction request, will add to the prediction request the client session ID. When the prediction response is received back by the application server, it will record both the model instance and the client session ID. An analysis program will then process these records to measure which model A or B better optimized the business objective.\n", + "\n", + "### Build the Wrapper Model\n", + "\n", + "Let's get started. We can do this in three lines of Keras code!\n", + "\n", + "1. Create a `Lambda()` layer. In this layer, you will take as input the softmax output from the model. You will then output the softmax output along with a numerical identifier representing the client session. Because this is a model, for the identifier, you need to:\n", + "\n", + "- Output as a number.\n", + "- Output the value as graph operator constant using `tf.constant()`.\n", + "- Give it a tensor shape (not-scalar), but specifing the value as a list and then convert to a tensor.\n", + "\n", + "2. Create a wrapper model around the original model, where:\n", + "\n", + "- The input is the original model input.\n", + "- The output is the Lambda layer.\n", + "\n", + "When you deploy the model, you will use the wrapper version instead of the original model." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ab_testing_wrapper" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "from tensorflow.keras import Input, Model\n", + "from tensorflow.keras.layers import Lambda\n", + "\n", + "softmax = model_A.outputs[0]\n", + "outputs = Lambda(lambda z: (z, tf.convert_to_tensor([tf.constant(0)])))(softmax)\n", + "wrapper_model_A = Model(model_A.inputs, outputs)\n", + "\n", + "softmax = model_B.outputs[0]\n", + "outputs = Lambda(lambda z: (z, tf.convert_to_tensor([tf.constant(1)])))(softmax)\n", + "wrapper_model_B = Model(model_B.inputs, outputs)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ab_testing_local_prediction" + }, + "source": [ + "#### Local Prediction\n", + "\n", + "Let's now do a local prediction with one of your wrapper A/B models. You will pass three instances (images) for prediction, and get back:\n", + "\n", + "- The softmax prediction for each instance request.\n", + "- The model A/B identifier. In this case 0 for A." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ab_testing_local_prediction" + }, + "outputs": [], + "source": [ + "wrapper_model_A.predict(x_test[0:3])" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_image" + }, + "source": [ + "### Serving function for image data\n", + "\n", + "To pass images to the prediction service, you encode the compressed (e.g., JPEG) image bytes into base 64 -- which makes the content safe from modification while transmitting binary data over the network. Since this deployed model expects input data as raw (uncompressed) bytes, you need to ensure that the base 64 encoded data gets converted back to raw bytes before it is passed as input to the deployed model.\n", + "\n", + "To resolve this, define a serving function (`serving_fn`) and attach it to the model as a preprocessing step. Add a `@tf.function` decorator so the serving function is fused to the underlying model (instead of upstream on a CPU).\n", + "\n", + "When you send a prediction or explanation request, the content of the request is base 64 decoded into a Tensorflow string (`tf.string`), which is passed to the serving function (`serving_fn`). The serving function preprocesses the `tf.string` into raw (uncompressed) numpy bytes (`preprocess_fn`) to match the input requirements of the model:\n", + "- `io.decode_jpeg`- Decompresses the JPG image which is returned as a Tensorflow tensor with three channels (RGB).\n", + "- `image.convert_image_dtype` - Changes integer pixel values to float 32.\n", + "- `image.resize` - Resizes the image to match the input shape for the model.\n", + "- `resized / 255.0` - Rescales (normalization) the pixel data between 0 and 1.\n", + "\n", + "At this point, the data can be passed to the model (`m_call`)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_image:ab_testing,a" + }, + "outputs": [], + "source": [ + "CONCRETE_INPUT = \"numpy_inputs\"\n", + "\n", + "\n", + "def _preprocess(bytes_input):\n", + " decoded = tf.io.decode_jpeg(bytes_input, channels=3)\n", + " decoded = tf.image.convert_image_dtype(decoded, tf.float32)\n", + " resized = tf.image.resize(decoded, size=(32, 32))\n", + " rescale = tf.cast(resized / 255.0, tf.float32)\n", + " return rescale\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def preprocess_fn(bytes_inputs):\n", + " decoded_images = tf.map_fn(\n", + " _preprocess, bytes_inputs, dtype=tf.float32, back_prop=False\n", + " )\n", + " return {\n", + " CONCRETE_INPUT: decoded_images\n", + " } # User needs to make sure the key matches model's input\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def serving_fn(bytes_inputs):\n", + " images = preprocess_fn(bytes_inputs)\n", + " prob = m_call(**images)\n", + " return prob\n", + "\n", + "\n", + "m_call = tf.function(wrapper_model_A.call).get_concrete_function(\n", + " [tf.TensorSpec(shape=[None, 32, 32, 3], dtype=tf.float32, name=CONCRETE_INPUT)]\n", + ")\n", + "\n", + "tf.saved_model.save(\n", + " wrapper_model_A, model_path_to_deploy_A, signatures={\"serving_default\": serving_fn}\n", + ")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_image:ab_testing,b" + }, + "outputs": [], + "source": [ + "CONCRETE_INPUT = \"numpy_inputs\"\n", + "\n", + "\n", + "def _preprocess(bytes_input):\n", + " decoded = tf.io.decode_jpeg(bytes_input, channels=3)\n", + " decoded = tf.image.convert_image_dtype(decoded, tf.float32)\n", + " resized = tf.image.resize(decoded, size=(32, 32))\n", + " rescale = tf.cast(resized / 255.0, tf.float32)\n", + " return rescale\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def preprocess_fn(bytes_inputs):\n", + " decoded_images = tf.map_fn(\n", + " _preprocess, bytes_inputs, dtype=tf.float32, back_prop=False\n", + " )\n", + " return {\n", + " CONCRETE_INPUT: decoded_images\n", + " } # User needs to make sure the key matches model's input\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def serving_fn(bytes_inputs):\n", + " images = preprocess_fn(bytes_inputs)\n", + " prob = m_call(**images)\n", + " return prob\n", + "\n", + "\n", + "m_call = tf.function(wrapper_model_B.call).get_concrete_function(\n", + " [tf.TensorSpec(shape=[None, 32, 32, 3], dtype=tf.float32, name=CONCRETE_INPUT)]\n", + ")\n", + "\n", + "tf.saved_model.save(\n", + " wrapper_model_B, model_path_to_deploy_B, signatures={\"serving_default\": serving_fn}\n", + ")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_signature:ab_testing" + }, + "outputs": [], + "source": [ + "loaded_A = tf.saved_model.load(model_path_to_deploy_A)\n", + "loaded_B = tf.saved_model.load(model_path_to_deploy_B)\n", + "\n", + "serving_input = list(\n", + " loaded_A.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "print(\"Serving function input:\", serving_input)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "upload_the_model" + }, + "source": [ + "### Upload the model\n", + "\n", + "Use this helper function `upload_model` to upload your model, stored in SavedModel format, up to the `Model` service, which will instantiate a Vertex `Model` resource instance for your model. Once you've done that, you can use the `Model` resource instance in the same way as any other Vertex `Model` resource instance, such as deploying to an `Endpoint` resource for serving predictions.\n", + "\n", + "The helper function takes the following parameters:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` service.\n", + "- `image_uri`: The container image for the model deployment.\n", + "- `model_uri`: The Cloud Storage path to our SavedModel artificat. For this tutorial, this is the Cloud Storage location where the `trainer/task.py` saved the model artifacts, which we specified in the variable `MODEL_DIR`.\n", + "\n", + "The helper function calls the `Model` client service's method `upload_model`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for `Dataset`, `Model` and `Endpoint` resources.\n", + "- `model`: The specification for the Vertex `Model` resource instance.\n", + "\n", + "Let's now dive deeper into the Vertex model specification `model`. This is a dictionary object that consists of the following fields:\n", + "\n", + "- `display_name`: A human readable name for the `Model` resource.\n", + "- `metadata_schema_uri`: Since your model was built without an Vertex `Dataset` resource, you will leave this blank (`''`).\n", + "- `artificat_uri`: The Cloud Storage path where the model is stored in SavedModel format.\n", + "- `container_spec`: This is the specification for the Docker container that will be installed on the `Endpoint` resource, from which the `Model` resource will serve predictions. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + "\n", + "Uploading a model into a Vertex Model resource returns a long running operation, since it may take a few moments. You call response.result(), which is a synchronous call and will return when the Vertex Model resource is ready.\n", + "\n", + "The helper function returns the Vertex fully qualified identifier for the corresponding Vertex Model instance upload_model_response.model. You will save the identifier for subsequent steps in the variable model_to_deploy_id." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "upload_the_model:ab_testing" + }, + "outputs": [], + "source": [ + "IMAGE_URI = DEPLOY_IMAGE\n", + "\n", + "\n", + "def upload_model(display_name, image_uri, model_uri):\n", + " model = {\n", + " \"display_name\": display_name,\n", + " \"metadata_schema_uri\": \"\",\n", + " \"artifact_uri\": model_uri,\n", + " \"container_spec\": {\n", + " \"image_uri\": image_uri,\n", + " \"command\": [],\n", + " \"args\": [],\n", + " \"env\": [{\"name\": \"env_name\", \"value\": \"env_value\"}],\n", + " \"ports\": [{\"container_port\": 8080}],\n", + " \"predict_route\": \"\",\n", + " \"health_route\": \"\",\n", + " },\n", + " }\n", + " response = clients[\"model\"].upload_model(parent=PARENT, model=model)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " upload_model_response = response.result(timeout=180)\n", + " print(\"upload_model_response\")\n", + " print(\" model:\", upload_model_response.model)\n", + " return upload_model_response.model\n", + "\n", + "\n", + "model_to_deploy_id_A = upload_model(\n", + " \"cifar10-\" + TIMESTAMP, IMAGE_URI, model_path_to_deploy_A\n", + ")\n", + "model_to_deploy_id_B = upload_model(\n", + " \"cifar10-\" + TIMESTAMP, IMAGE_URI, model_path_to_deploy_B\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:custom" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex custom `Model` resource. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"cifar10_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:dedicated" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `dedicated_resources`: This refers to how many compute instances (replicas) that are scaled for serving prediction requests.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `min_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:dedicated,ab_testing" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"cifar10_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " # Accelerators can be used only if the model specifies a GPU image.\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " \"machine_spec\": machine_spec,\n", + " },\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id_A = deploy_model(\n", + " model_to_deploy_id_A, DEPLOYED_NAME + \"-A\", endpoint_id, {\"0\": 100}\n", + ")\n", + "deployed_model_id_B = deploy_model(\n", + " model_to_deploy_id_B,\n", + " DEPLOYED_NAME + \"-B\",\n", + " endpoint_id,\n", + " {\"0\": 50, deployed_model_id_A: 50},\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_item:test" + }, + "source": [ + "### Get test item\n", + "\n", + "You will use an example out of the test (holdout) portion of the dataset as a test item." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:test" + }, + "outputs": [], + "source": [ + "test_image = x_test[0]\n", + "test_label = y_test[0]\n", + "print(test_image.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "prepare_test_item:test,image" + }, + "source": [ + "### Prepare the request content\n", + "You are going to send the CIFAR10 image as compressed JPG image, instead of the raw uncompressed bytes:\n", + "\n", + "- `cv2.imwrite`: Use openCV to write the uncompressed image to disk as a compressed JPEG image.\n", + " - Denormalize the image data from \\[0,1) range back to [0,255).\n", + " - Convert the 32-bit floating point values to 8-bit unsigned integers.\n", + "- `tf.io.read_file`: Read the compressed JPG images back into memory as raw bytes.\n", + "- `base64.b64encode`: Encode the raw bytes into a base 64 encoded string." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "prepare_test_item:test,image" + }, + "outputs": [], + "source": [ + "import base64\n", + "\n", + "import cv2\n", + "\n", + "cv2.imwrite(\"tmp.jpg\", (test_image * 255).astype(np.uint8))\n", + "\n", + "bytes = tf.io.read_file(\"tmp.jpg\")\n", + "b64str = base64.b64encode(bytes.numpy()).decode(\"utf-8\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "send_prediction_request:image,ab_testing" + }, + "source": [ + "### Send the prediction request\n", + "\n", + "Ok, now you have a test image. Use this helper function `predict_image`, which takes the following parameters:\n", + "\n", + "- `image`: The test image data as a numpy array.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed to.\n", + "- `parameters_dict`: Additional parameters for serving.\n", + "\n", + "This function calls the prediction client service `predict` method with the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed to.\n", + "- `instances`: A list of instances (encoded images) to predict.\n", + "- `parameters`: Additional parameters for serving.\n", + "\n", + "To pass the image data to the prediction service, in the previous step you encoded the bytes into base64 -- which makes the content safe from modification when transmitting binary data over the network. You need to tell the serving binary where your model is deployed to, that the content has been base64 encoded, so it will decode it on the other end in the serving binary.\n", + "\n", + "Each instance in the prediction request is a dictionary entry of the form:\n", + "\n", + " {serving_input: {'b64': content}}\n", + "\n", + "- `input_name`: the name of the input layer of the underlying model.\n", + "- `'b64'`: A key that indicates the content is base64 encoded.\n", + "- `content`: The compressed JPG image bytes as a base64 encoded string.\n", + "\n", + "Since the `predict()` service can take multiple images (instances), you will send your single image as a list of one image. As a final step, you package the instances list into Google's protobuf format -- which is what we pass to the `predict()` service.\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding image in the request. You will see in the output for each prediction:\n", + "\n", + "- `predictions`: Confidence level for the prediction, between 0 and 1, for each of the classes.\n", + "- `output_2`: The client session ID." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "send_prediction_request:image,ab_testing" + }, + "outputs": [], + "source": [ + "def predict_image(image, endpoint, parameters_dict):\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [{serving_input: {\"b64\": image}}]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters_dict\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", dict(prediction))\n", + "\n", + "\n", + "predict_image(b64str, endpoint_id, None)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model:ab_testing" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id_A, endpoint_id)\n", + "undeploy_model(deployed_model_id_B, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_custom_image_classification_online_ab_testing.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_custom_image_classification_online_container.ipynb b/notebooks/community/gapic/custom/showcase_custom_image_classification_online_container.ipynb new file mode 100644 index 000000000..74982e607 --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_custom_image_classification_online_container.ipynb @@ -0,0 +1,2294 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Custom training image classification model with custom container for online prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom,training,custom_container" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to train using a custom container and deploy a custom image classification model for online prediction." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:custom,cifar10,icn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [CIFAR10 dataset](https://www.tensorflow.org/datasets/catalog/cifar10) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use is built into TensorFlow. The trained model predicts which type of class an image is from ten classes: airplane, automobile, bird, cat, deer, dog, frog, horse, ship, truck." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,training,custom_container,online_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create a custom model from a Python script in a custom Docker container using the Vertex client library, and then do a prediction on the deployed model by sending data. You can alternatively create custom models using `gcloud` command-line tool or online using Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex custom job for training a model.\n", + "- Train a TensorFlow model using a custom container.\n", + "- Retrieve and load the model artifacts.\n", + "- View the model evaluation.\n", + "- Upload the model as a Vertex `Model` resource.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for training and prediction.\n", + "\n", + "Set the variables `TRAIN_GPU/TRAIN_NGPU` and `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU.\n", + "\n", + "*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3 -- which is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n", + " TRAIN_GPU, TRAIN_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n", + " )\n", + "else:\n", + " TRAIN_GPU, TRAIN_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (None, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:prediction" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for prediction\n", + "\n", + "- Set the variable `TF` to the TensorFlow version of the container image. For example, `2-1` would be version 2.1, and `1-15` would be version 1.15. The following list shows some of the pre-built images available:\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-3:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-90:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-82:latest`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-23:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-22:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-20:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for prediction](https://cloud.google.com/vertex-ai/docs/predictions/pre-built-containers)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf2-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf2-cpu.{}\".format(TF)\n", + "else:\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "DEPLOY_IMAGE = \"gcr.io/cloud-aiplatform/prediction/{}:latest\".format(DEPLOY_VERSION)\n", + "\n", + "print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:training,prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for training and prediction.\n", + "\n", + "- Set the variables `TRAIN_COMPUTE` and `DEPLOY_COMPUTE` to configure the compute resources for the VMs you will use for for training and prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: The following is not supported for training:*\n", + "\n", + " - `standard`: 2 vCPUs\n", + " - `highcpu`: 2, 4 and 8 vCPUs\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_TRAIN_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Train machine type\", TRAIN_COMPUTE)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own custom model and training for CIFAR10." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:custom" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Model Service for `Model` resources.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:custom" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"job\"] = create_job_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_model" + }, + "source": [ + "## Train a model\n", + "\n", + "There are two ways you can train a custom model using a container image:\n", + "\n", + "- **Use a Google Cloud prebuilt container**. If you use a prebuilt container, you will additionally specify a Python package to install into the container image. This Python package contains your code for training a custom model.\n", + "\n", + "- **Use your own custom container image**. If you use your own container, the container needs to contain your code for training a custom model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_docker_container:training" + }, + "source": [ + "### Create a Docker file\n", + "\n", + "In this tutorial, you train a CIFAR10 model using your own custom container.\n", + "\n", + "To use your own custom container, you build a Docker file. First, you will create a directory for the container components." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "examine_training_package" + }, + "source": [ + "### Examine the training package\n", + "\n", + "#### Package layout\n", + "\n", + "Before you start the training, you will look at how a Python package is assembled for a custom training job. When unarchived, the package contains the following directory/file layout.\n", + "\n", + "- PKG-INFO\n", + "- README.md\n", + "- setup.cfg\n", + "- setup.py\n", + "- trainer\n", + " - \\_\\_init\\_\\_.py\n", + " - task.py\n", + "\n", + "The files `setup.cfg` and `setup.py` are the instructions for installing the package into the operating environment of the Docker image.\n", + "\n", + "The file `trainer/task.py` is the Python script for executing the custom training job. *Note*, when we referred to it in the worker pool specification, we replace the directory slash with a dot (`trainer.task`) and dropped the file suffix (`.py`).\n", + "\n", + "#### Package Assembly\n", + "\n", + "In the following cells, you will assemble the training package." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "examine_training_package" + }, + "outputs": [], + "source": [ + "# Make folder for Python training script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\ntag_build =\\n\\ntag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'tensorflow_datasets==1.3.0',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\nName: CIFAR10 image classification\\n\\nVersion: 0.0.0\\n\\nSummary: Demostration training script\\n\\nHome-page: www.google.com\\n\\nAuthor: Google\\n\\nAuthor-email: aferlitsch@google.com\\n\\nLicense: Public\\n\\nDescription: Demo\\n\\nPlatform: Vertex\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:cifar10" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the training script task.py. We won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n", + "- Loads CIFAR10 dataset from TF Datasets (tfds).\n", + "- Builds a model using TF.Keras model API.\n", + "- Compiles the model (`compile()`).\n", + "- Sets a training distribution strategy according to the argument `args.distribute`.\n", + "- Trains the model (`fit()`) with epochs and steps according to the arguments `args.epochs` and `args.steps`\n", + "- Saves the trained model (`save(args.model_dir)`) to the specified model directory." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:cifar10" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Single, Mirror and Multi-Machine Distributed Training for CIFAR-10\n", + "\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "import argparse\n", + "import os\n", + "import sys\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv(\"AIP_MODEL_DIR\"), type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.01, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=10, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=200, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "print('DEVICES', device_lib.list_local_devices())\n", + "\n", + "# Single Machine, single compute device\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "# Single Machine, multiple compute device\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "# Multiple Machine, multiple compute device\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "# Multi-worker configuration\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "# Preparing dataset\n", + "BUFFER_SIZE = 10000\n", + "BATCH_SIZE = 64\n", + "\n", + "\n", + "def make_datasets_unbatched():\n", + "\n", + " # Scaling CIFAR10 data from (0, 255] to (0., 1.]\n", + " def scale(image, label):\n", + " image = tf.cast(image, tf.float32)\n", + " image /= 255.0\n", + " return image, label\n", + "\n", + "\n", + " datasets, info = tfds.load(name='cifar10',\n", + " with_info=True,\n", + " as_supervised=True)\n", + " return datasets['train'].map(scale).cache().shuffle(BUFFER_SIZE).repeat()\n", + "\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_cnn_model():\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu', input_shape=(32, 32, 3)),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu'),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Flatten(),\n", + " tf.keras.layers.Dense(10, activation='softmax')\n", + " ])\n", + " model.compile(\n", + " loss=tf.keras.losses.sparse_categorical_crossentropy,\n", + " optimizer=tf.keras.optimizers.SGD(learning_rate=args.lr),\n", + " metrics=['accuracy'])\n", + " return model\n", + "\n", + "\n", + "# Train the model\n", + "NUM_WORKERS = strategy.num_replicas_in_sync\n", + "# Here the batch size scales up by number of workers since\n", + "# `tf.data.Dataset.batch` expects the global batch size.\n", + "GLOBAL_BATCH_SIZE = BATCH_SIZE * NUM_WORKERS\n", + "train_dataset = make_datasets_unbatched().batch(GLOBAL_BATCH_SIZE)\n", + "\n", + "with strategy.scope():\n", + " # Creation of dataset, and model building/compiling need to be within\n", + " # `strategy.scope()`.\n", + " model = build_and_compile_cnn_model()\n", + "\n", + "model.fit(x=train_dataset, epochs=args.epochs, steps_per_epoch=args.steps)\n", + "model.save(args.model_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "write_docker_file:training,tf-dlvm" + }, + "source": [ + "#### Write the Docker file contents\n", + "\n", + "Your first step in containerizing your code is to create a Docker file. In your Docker you’ll include all the commands needed to run your container image. It’ll install all the libraries you’re using and set up the entry point for your training code.\n", + "\n", + "1. Install a pre-defined container image from TensorFlow repository for deep learning images.\n", + "2. Copies in the Python training code, to be shown subsequently.\n", + "3. Sets the entry into the Python training script as `trainer/task.py`. Note, the `.py` is dropped in the ENTRYPOINT command, as it is implied." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "write_docker_file:training,tf-dlvm" + }, + "outputs": [], + "source": [ + "%%writefile custom/Dockerfile\n", + "\n", + "FROM gcr.io/deeplearning-platform-release/tf2-cpu.2-1\n", + "WORKDIR /root\n", + "\n", + "WORKDIR /\n", + "\n", + "# Copies the trainer code to the docker image.\n", + "COPY trainer /trainer\n", + "\n", + "# Sets up the entry point to invoke the trainer.\n", + "ENTRYPOINT [\"python\", \"-m\", \"trainer.task\"]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "name_container:training" + }, + "source": [ + "#### Build the container locally\n", + "\n", + "Next, you will provide a name for your customer container that you will use when you submit it to the Google Container Registry." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "name_container:training" + }, + "outputs": [], + "source": [ + "TRAIN_IMAGE = \"gcr.io/\" + PROJECT_ID + \"/cifar10:v1\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "build_container:training" + }, + "source": [ + "Next, build the container." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "build_container:training" + }, + "outputs": [], + "source": [ + "! docker build custom -t $TRAIN_IMAGE" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "test_container:training" + }, + "source": [ + "#### Test the container locally\n", + "\n", + "Run the container within your notebook instance to ensure it’s working correctly. You will run it for 5 epochs." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "test_container:training" + }, + "outputs": [], + "source": [ + "! docker run $TRAIN_IMAGE --epochs=5" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "register_container:training" + }, + "source": [ + "#### Register the custom container\n", + "\n", + "When you’ve finished running the container locally, push it to Google Container Registry." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "register_container:training" + }, + "outputs": [], + "source": [ + "! docker push $TRAIN_IMAGE" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tarball_training_script" + }, + "source": [ + "#### Store training script on your Cloud Storage bucket\n", + "\n", + "Next, you package the training folder into a compressed tar ball, and then store it in your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tarball_training_script" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_cifar10.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_specification:custom_container" + }, + "source": [ + "## Prepare your custom job specification\n", + "\n", + "Now that your clients are ready, your first step is to create a Job Specification for your custom training job. The job specification will consist of the following:\n", + "\n", + "- `worker_pool_spec` : The specification of the type of machine(s) you will use for training and how many (single or distributed)\n", + "- `container_spec` : The specification of the custom container." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "source": [ + "### Prepare your machine specification\n", + "\n", + "Now define the machine specification for your custom training job. This tells Vertex what type of machine instance to provision for the training.\n", + " - `machine_type`: The type of GCP instance to provision -- e.g., n1-standard-8.\n", + " - `accelerator_type`: The type, if any, of hardware accelerator. In this tutorial if you previously set the variable `TRAIN_GPU != None`, you are using a GPU; otherwise you will use a CPU.\n", + " - `accelerator_count`: The number of accelerators." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "outputs": [], + "source": [ + "if TRAIN_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": TRAIN_COMPUTE,\n", + " \"accelerator_type\": TRAIN_GPU,\n", + " \"accelerator_count\": TRAIN_NGPU,\n", + " }\n", + "else:\n", + " machine_spec = {\"machine_type\": TRAIN_COMPUTE, \"accelerator_count\": 0}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "source": [ + "### Prepare your disk specification\n", + "\n", + "(optional) Now define the disk specification for your custom training job. This tells Vertex what type and size of disk to provision in each machine instance for the training.\n", + "\n", + " - `boot_disk_type`: Either SSD or Standard. SSD is faster, and Standard is less expensive. Defaults to SSD.\n", + " - `boot_disk_size_gb`: Size of disk in GB." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "outputs": [], + "source": [ + "DISK_TYPE = \"pd-ssd\" # [ pd-ssd, pd-standard]\n", + "DISK_SIZE = 200 # GB\n", + "\n", + "disk_spec = {\"boot_disk_type\": DISK_TYPE, \"boot_disk_size_gb\": DISK_SIZE}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "prepare_custom_container_specification" + }, + "source": [ + "### Prepare your container specification\n", + "\n", + "Now define the container specification for your custom training container:\n", + "\n", + "- `image_uri`: The custom container image.\n", + "- `args`: The command-line arguments to pass to the executable that is set as the entry point into the container.\n", + " - `--model-dir` : For our demonstrations, we use this command-line argument to specify where to store the model artifacts.\n", + " - direct: You pass the Cloud Storage location as a command line argument to your training script (set variable `DIRECT = True`), or\n", + " - indirect: The service passes the Cloud Storage location as the environment variable `AIP_MODEL_DIR` to your training script (set variable `DIRECT = False`). In this case, you tell the service the model artifact location in the job specification.\n", + " - `\"--epochs=\" + EPOCHS`: The number of epochs for training.\n", + " - `\"--steps=\" + STEPS`: The number of steps per epoch." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "prepare_custom_container_specification" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"_custom_container\" + TIMESTAMP\n", + "MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, JOB_NAME)\n", + "\n", + "EPOCHS = 20\n", + "STEPS = 100\n", + "\n", + "DIRECT = True\n", + "if DIRECT:\n", + " CMDARGS = [\n", + " \"--model-dir=\" + MODEL_DIR,\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " ]\n", + "else:\n", + " CMDARGS = [\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " ]\n", + "\n", + "container_spec = {\n", + " \"image_uri\": TRAIN_IMAGE,\n", + " \"args\": CMDARGS,\n", + "}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_worker_pool_specification:custom_container" + }, + "source": [ + "### Define the worker pool specification\n", + "\n", + "Next, you define the worker pool specification for your custom training job. The worker pool specification will consist of the following:\n", + "\n", + "- `replica_count`: The number of instances to provision of this machine type.\n", + "- `machine_spec`: The hardware specification.\n", + "- `disk_spec` : (optional) The disk storage specification.\n", + "- `container_spec`: The Docker container to install on the VM instance(s)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_worker_pool_specification:custom_container" + }, + "outputs": [], + "source": [ + "worker_pool_spec = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": machine_spec,\n", + " \"container_spec\": container_spec,\n", + " \"disk_spec\": disk_spec,\n", + " }\n", + "]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "assemble_custom_job_specification" + }, + "source": [ + "### Assemble a job specification\n", + "\n", + "Now assemble the complete description for the custom job specification:\n", + "\n", + "- `display_name`: The human readable name you assign to this custom job.\n", + "- `job_spec`: The specification for the custom job.\n", + " - `worker_pool_specs`: The specification for the machine VM instances.\n", + " - `base_output_directory`: This tells the service the Cloud Storage location where to save the model artifacts (when variable `DIRECT = False`). The service will then pass the location to the training script as the environment variable `AIP_MODEL_DIR`, and the path will be of the form:\n", + "\n", + " /model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "assemble_custom_job_specification" + }, + "outputs": [], + "source": [ + "if DIRECT:\n", + " job_spec = {\"worker_pool_specs\": worker_pool_spec}\n", + "else:\n", + " job_spec = {\n", + " \"worker_pool_specs\": worker_pool_spec,\n", + " \"base_output_directory\": {\"output_uri_prefix\": MODEL_DIR},\n", + " }\n", + "\n", + "custom_job = {\"display_name\": JOB_NAME, \"job_spec\": job_spec}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job" + }, + "source": [ + "### Train the model\n", + "\n", + "\n", + "Now start the training of your custom training job on Vertex. Use this helper function `create_custom_job`, which takes the following parameter:\n", + "\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "The helper function calls job client service's `create_custom_job` method, with the following parameters:\n", + "\n", + "-`parent`: The Vertex location path to `Dataset`, `Model` and `Endpoint` resources.\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "You will display a handful of the fields returned in `response` object, with the two that are of most interest are:\n", + "\n", + "`response.name`: The Vertex fully qualified identifier assigned to this custom training job. You save this identifier for using in subsequent steps.\n", + "\n", + "`response.state`: The current state of the custom training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job" + }, + "outputs": [], + "source": [ + "def create_custom_job(custom_job):\n", + " response = clients[\"job\"].create_custom_job(parent=PARENT, custom_job=custom_job)\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = create_custom_job(custom_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "job_id:response" + }, + "source": [ + "Now get the unique identifier for the custom job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom job\n", + "job_id = response.name\n", + "# The short numeric ID for the custom job\n", + "job_short_id = job_id.split(\"/\")[-1]\n", + "\n", + "print(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_custom_job" + }, + "source": [ + "### Get information on a custom job\n", + "\n", + "Next, use this helper function `get_custom_job`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "The helper function calls the job client service's`get_custom_job` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "If you recall, you got the Vertex fully qualified identifier for the custom job in the `response.name` field when you called the `create_custom_job` method, and saved the identifier in the variable `job_id`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_custom_job" + }, + "outputs": [], + "source": [ + "def get_custom_job(name, silent=False):\n", + " response = clients[\"job\"].get_custom_job(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = get_custom_job(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete:custom" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, we will need to know the location of the saved model, which the Python script saved in your local Cloud Storage bucket at `MODEL_DIR + '/saved_model.pb'`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete:custom" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_custom_job(job_id, True)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_path_to_deploy = None\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " if not DIRECT:\n", + " MODEL_DIR = MODEL_DIR + \"/model\"\n", + " model_path_to_deploy = MODEL_DIR\n", + " print(\"Training Time:\", response.update_time - response.create_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model_to_deploy:\", model_path_to_deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "load_saved_model" + }, + "source": [ + "## Load the saved model\n", + "\n", + "Your model is stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Now load it from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n", + "\n", + "To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "load_saved_model" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:image" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load the CIFAR10 test (holdout) data from `tf.keras.datasets`, using the method `load_data()`. This will return the dataset as a tuple of two elements. The first element is the training data and the second is the test data. Each element is also a tuple of two elements: the image data, and the corresponding labels.\n", + "\n", + "You don't need the training data, and hence why we loaded it as `(_, _)`.\n", + "\n", + "Before you can run the data through evaluation, you need to preprocess it:\n", + "\n", + "x_test:\n", + "1. Normalize (rescaling) the pixel data by dividing each pixel by 255. This will replace each single byte integer pixel with a 32-bit floating point number between 0 and 1.\n", + "\n", + "y_test:
\n", + "2. The labels are currently scalar (sparse). If you look back at the `compile()` step in the `trainer/task.py` script, you will find that it was compiled for sparse labels. So we don't need to do anything more." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "evaluate_custom_model:image,cifar10" + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "from tensorflow.keras.datasets import cifar10\n", + "\n", + "(_, _), (x_test, y_test) = cifar10.load_data()\n", + "x_test = (x_test / 255.0).astype(np.float32)\n", + "\n", + "print(x_test.shape, y_test.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "perform_evaluation_custom" + }, + "source": [ + "### Perform the model evaluation\n", + "\n", + "Now evaluate how well the model in the custom job did." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "perform_evaluation_custom" + }, + "outputs": [], + "source": [ + "model.evaluate(x_test, y_test)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "how_serving_function_works" + }, + "source": [ + "## Upload the model for serving\n", + "\n", + "Next, you will upload your TF.Keras model from the custom job to Vertex `Model` service, which will create a Vertex `Model` resource for your custom model. During upload, you need to define a serving function to convert data to the format your model expects. If you send encoded data to Vertex, your serving function ensures that the data is decoded on the model server before it is passed as input to your model.\n", + "\n", + "### How does the serving function work\n", + "\n", + "When you send a request to an online prediction server, the request is received by a HTTP server. The HTTP server extracts the prediction request from the HTTP request content body. The extracted prediction request is forwarded to the serving function. For Google pre-built prediction containers, the request content is passed to the serving function as a `tf.string`.\n", + "\n", + "The serving function consists of two parts:\n", + "\n", + "- `preprocessing function`:\n", + " - Converts the input (`tf.string`) to the input shape and data type of the underlying model (dynamic graph).\n", + " - Performs the same preprocessing of the data that was done during training the underlying model -- e.g., normalizing, scaling, etc.\n", + "- `post-processing function`:\n", + " - Converts the model output to format expected by the receiving application -- e.q., compresses the output.\n", + " - Packages the output for the the receiving application -- e.g., add headings, make JSON object, etc.\n", + "\n", + "Both the preprocessing and post-processing functions are converted to static graphs which are fused to the model. The output from the underlying model is passed to the post-processing function. The post-processing function passes the converted/packaged output back to the HTTP server. The HTTP server returns the output as the HTTP response content.\n", + "\n", + "One consideration you need to consider when building serving functions for TF.Keras models is that they run as static graphs. That means, you cannot use TF graph operations that require a dynamic graph. If you do, you will get an error during the compile of the serving function which will indicate that you are using an EagerTensor which is not supported." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_image" + }, + "source": [ + "### Serving function for image data\n", + "\n", + "To pass images to the prediction service, you encode the compressed (e.g., JPEG) image bytes into base 64 -- which makes the content safe from modification while transmitting binary data over the network. Since this deployed model expects input data as raw (uncompressed) bytes, you need to ensure that the base 64 encoded data gets converted back to raw bytes before it is passed as input to the deployed model.\n", + "\n", + "To resolve this, define a serving function (`serving_fn`) and attach it to the model as a preprocessing step. Add a `@tf.function` decorator so the serving function is fused to the underlying model (instead of upstream on a CPU).\n", + "\n", + "When you send a prediction or explanation request, the content of the request is base 64 decoded into a Tensorflow string (`tf.string`), which is passed to the serving function (`serving_fn`). The serving function preprocesses the `tf.string` into raw (uncompressed) numpy bytes (`preprocess_fn`) to match the input requirements of the model:\n", + "- `io.decode_jpeg`- Decompresses the JPG image which is returned as a Tensorflow tensor with three channels (RGB).\n", + "- `image.convert_image_dtype` - Changes integer pixel values to float 32.\n", + "- `image.resize` - Resizes the image to match the input shape for the model.\n", + "- `resized / 255.0` - Rescales (normalization) the pixel data between 0 and 1.\n", + "\n", + "At this point, the data can be passed to the model (`m_call`)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_image" + }, + "outputs": [], + "source": [ + "CONCRETE_INPUT = \"numpy_inputs\"\n", + "\n", + "\n", + "def _preprocess(bytes_input):\n", + " decoded = tf.io.decode_jpeg(bytes_input, channels=3)\n", + " decoded = tf.image.convert_image_dtype(decoded, tf.float32)\n", + " resized = tf.image.resize(decoded, size=(32, 32))\n", + " rescale = tf.cast(resized / 255.0, tf.float32)\n", + " return rescale\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def preprocess_fn(bytes_inputs):\n", + " decoded_images = tf.map_fn(\n", + " _preprocess, bytes_inputs, dtype=tf.float32, back_prop=False\n", + " )\n", + " return {\n", + " CONCRETE_INPUT: decoded_images\n", + " } # User needs to make sure the key matches model's input\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def serving_fn(bytes_inputs):\n", + " images = preprocess_fn(bytes_inputs)\n", + " prob = m_call(**images)\n", + " return prob\n", + "\n", + "\n", + "m_call = tf.function(model.call).get_concrete_function(\n", + " [tf.TensorSpec(shape=[None, 32, 32, 3], dtype=tf.float32, name=CONCRETE_INPUT)]\n", + ")\n", + "\n", + "tf.saved_model.save(\n", + " model, model_path_to_deploy, signatures={\"serving_default\": serving_fn}\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_signature:image" + }, + "source": [ + "## Get the serving function signature\n", + "\n", + "You can get the signatures of your model's input and output layers by reloading the model into memory, and querying it for the signatures corresponding to each layer.\n", + "\n", + "For your purpose, you need the signature of the serving function. Why? Well, when we send our data for prediction as a HTTP request packet, the image data is base64 encoded, and our TF.Keras model takes numpy input. Your serving function will do the conversion from base64 to a numpy array.\n", + "\n", + "When making a prediction request, you need to route the request to the serving function instead of the model, so you need to know the input layer name of the serving function -- which you will use later when you make a prediction request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_signature:image" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_path_to_deploy)\n", + "\n", + "serving_input = list(\n", + " loaded.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "print(\"Serving function input:\", serving_input)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "upload_the_model" + }, + "source": [ + "### Upload the model\n", + "\n", + "Use this helper function `upload_model` to upload your model, stored in SavedModel format, up to the `Model` service, which will instantiate a Vertex `Model` resource instance for your model. Once you've done that, you can use the `Model` resource instance in the same way as any other Vertex `Model` resource instance, such as deploying to an `Endpoint` resource for serving predictions.\n", + "\n", + "The helper function takes the following parameters:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` service.\n", + "- `image_uri`: The container image for the model deployment.\n", + "- `model_uri`: The Cloud Storage path to our SavedModel artificat. For this tutorial, this is the Cloud Storage location where the `trainer/task.py` saved the model artifacts, which we specified in the variable `MODEL_DIR`.\n", + "\n", + "The helper function calls the `Model` client service's method `upload_model`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for `Dataset`, `Model` and `Endpoint` resources.\n", + "- `model`: The specification for the Vertex `Model` resource instance.\n", + "\n", + "Let's now dive deeper into the Vertex model specification `model`. This is a dictionary object that consists of the following fields:\n", + "\n", + "- `display_name`: A human readable name for the `Model` resource.\n", + "- `metadata_schema_uri`: Since your model was built without an Vertex `Dataset` resource, you will leave this blank (`''`).\n", + "- `artificat_uri`: The Cloud Storage path where the model is stored in SavedModel format.\n", + "- `container_spec`: This is the specification for the Docker container that will be installed on the `Endpoint` resource, from which the `Model` resource will serve predictions. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + "\n", + "Uploading a model into a Vertex Model resource returns a long running operation, since it may take a few moments. You call response.result(), which is a synchronous call and will return when the Vertex Model resource is ready.\n", + "\n", + "The helper function returns the Vertex fully qualified identifier for the corresponding Vertex Model instance upload_model_response.model. You will save the identifier for subsequent steps in the variable model_to_deploy_id." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "upload_the_model" + }, + "outputs": [], + "source": [ + "IMAGE_URI = DEPLOY_IMAGE\n", + "\n", + "\n", + "def upload_model(display_name, image_uri, model_uri):\n", + " model = {\n", + " \"display_name\": display_name,\n", + " \"metadata_schema_uri\": \"\",\n", + " \"artifact_uri\": model_uri,\n", + " \"container_spec\": {\n", + " \"image_uri\": image_uri,\n", + " \"command\": [],\n", + " \"args\": [],\n", + " \"env\": [{\"name\": \"env_name\", \"value\": \"env_value\"}],\n", + " \"ports\": [{\"container_port\": 8080}],\n", + " \"predict_route\": \"\",\n", + " \"health_route\": \"\",\n", + " },\n", + " }\n", + " response = clients[\"model\"].upload_model(parent=PARENT, model=model)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " upload_model_response = response.result(timeout=180)\n", + " print(\"upload_model_response\")\n", + " print(\" model:\", upload_model_response.model)\n", + " return upload_model_response.model\n", + "\n", + "\n", + "model_to_deploy_id = upload_model(\n", + " \"cifar10-\" + TIMESTAMP, IMAGE_URI, model_path_to_deploy\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_model" + }, + "source": [ + "### Get `Model` resource information\n", + "\n", + "Now let's get the model information for just your model. Use this helper function `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource.\n", + "\n", + "This helper function calls the Vertex `Model` client service's method `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_model" + }, + "outputs": [], + "source": [ + "def get_model(name):\n", + " response = clients[\"model\"].get_model(name=name)\n", + " print(response)\n", + "\n", + "\n", + "get_model(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:custom" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex custom `Model` resource. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"cifar10_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:dedicated" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `dedicated_resources`: This refers to how many compute instances (replicas) that are scaled for serving prediction requests.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `min_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:dedicated" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"cifar10_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " \"machine_spec\": machine_spec,\n", + " },\n", + " \"disable_container_logging\": False,\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_item:test" + }, + "source": [ + "### Get test item\n", + "\n", + "You will use an example out of the test (holdout) portion of the dataset as a test item." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:test" + }, + "outputs": [], + "source": [ + "test_image = x_test[0]\n", + "test_label = y_test[0]\n", + "print(test_image.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "prepare_test_item:test,image" + }, + "source": [ + "### Prepare the request content\n", + "You are going to send the CIFAR10 image as compressed JPG image, instead of the raw uncompressed bytes:\n", + "\n", + "- `cv2.imwrite`: Use openCV to write the uncompressed image to disk as a compressed JPEG image.\n", + " - Denormalize the image data from \\[0,1) range back to [0,255).\n", + " - Convert the 32-bit floating point values to 8-bit unsigned integers.\n", + "- `tf.io.read_file`: Read the compressed JPG images back into memory as raw bytes.\n", + "- `base64.b64encode`: Encode the raw bytes into a base 64 encoded string." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "prepare_test_item:test,image" + }, + "outputs": [], + "source": [ + "import base64\n", + "\n", + "import cv2\n", + "\n", + "cv2.imwrite(\"tmp.jpg\", (test_image * 255).astype(np.uint8))\n", + "\n", + "bytes = tf.io.read_file(\"tmp.jpg\")\n", + "b64str = base64.b64encode(bytes.numpy()).decode(\"utf-8\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "send_prediction_request:image" + }, + "source": [ + "### Send the prediction request\n", + "\n", + "Ok, now you have a test image. Use this helper function `predict_image`, which takes the following parameters:\n", + "\n", + "- `image`: The test image data as a numpy array.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed to.\n", + "- `parameters_dict`: Additional parameters for serving.\n", + "\n", + "This function calls the prediction client service `predict` method with the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed to.\n", + "- `instances`: A list of instances (encoded images) to predict.\n", + "- `parameters`: Additional parameters for serving.\n", + "\n", + "To pass the image data to the prediction service, in the previous step you encoded the bytes into base64 -- which makes the content safe from modification when transmitting binary data over the network. You need to tell the serving binary where your model is deployed to, that the content has been base64 encoded, so it will decode it on the other end in the serving binary.\n", + "\n", + "Each instance in the prediction request is a dictionary entry of the form:\n", + "\n", + " {serving_input: {'b64': content}}\n", + "\n", + "- `input_name`: the name of the input layer of the underlying model.\n", + "- `'b64'`: A key that indicates the content is base64 encoded.\n", + "- `content`: The compressed JPG image bytes as a base64 encoded string.\n", + "\n", + "Since the `predict()` service can take multiple images (instances), you will send your single image as a list of one image. As a final step, you package the instances list into Google's protobuf format -- which is what we pass to the `predict()` service.\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding image in the request. You will see in the output for each prediction:\n", + "\n", + "- `predictions`: Confidence level for the prediction, between 0 and 1, for each of the classes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "send_prediction_request:image" + }, + "outputs": [], + "source": [ + "def predict_image(image, endpoint, parameters_dict):\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [{serving_input: {\"b64\": image}}]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters_dict\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", prediction)\n", + "\n", + "\n", + "predict_image(b64str, endpoint_id, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_custom_image_classification_online_container.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_custom_image_classification_online_explain.ipynb b/notebooks/community/gapic/custom/showcase_custom_image_classification_online_explain.ipynb new file mode 100644 index 000000000..5dbe91da9 --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_custom_image_classification_online_explain.ipynb @@ -0,0 +1,2514 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Custom training image classification model for online prediction with explanation\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom,xai" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to train and deploy a custom image classification model for online prediction with explanation." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:custom,cifar10,icn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [CIFAR10 dataset](https://www.tensorflow.org/datasets/catalog/cifar10) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use is built into TensorFlow. The trained model predicts which type of class an image is from ten classes: airplane, automobile, bird, cat, deer, dog, frog, horse, ship, truck." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,training,online_prediction,xai" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create a custom model from a Python script in a Google prebuilt Docker container using the Vertex client library, and then do a prediction with explanations on the deployed model by sending data. You can alternatively create custom models using `gcloud` command-line tool or online using Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex custom job for training a model.\n", + "- Train a TensorFlow model.\n", + "- Retrieve and load the model artifacts.\n", + "- View the model evaluation.\n", + "- Set explanation parameters.\n", + "- Upload the model as a Vertex `Model` resource.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction with explanation.\n", + "- Undeploy the `Model` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_tensorflow" + }, + "outputs": [], + "source": [ + "if os.environ[\"IS_TESTING\"]:\n", + " ! pip3 install -U tensorflow" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_cv2" + }, + "outputs": [], + "source": [ + "if os.environ[\"IS_TESTING\"]:\n", + " ! pip3 install -U opencv-python" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:v1beta1,protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:v1beta1,protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "import google.cloud.aiplatform_v1beta1 as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for training and prediction.\n", + "\n", + "Set the variables `TRAIN_GPU/TRAIN_NGPU` and `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU.\n", + "\n", + "*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3 -- which is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n", + " TRAIN_GPU, TRAIN_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n", + " )\n", + "else:\n", + " TRAIN_GPU, TRAIN_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (None, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:training,prediction" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for training and prediction\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-3:latest`\n", + " - TensorFlow 2.4\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-4:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/training/xgboost-cpu.1-1`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/training/scikit-learn-cpu.0-23:latest`\n", + " - Pytorch\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-5:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-6:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-7:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for training](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers).\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-3:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-90:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-82:latest`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-23:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-22:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-20:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for prediction](https://cloud.google.com/vertex-ai/docs/predictions/pre-built-containers)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf2-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf2-cpu.{}\".format(TF)\n", + "else:\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "TRAIN_IMAGE = \"gcr.io/cloud-aiplatform/training/{}:latest\".format(TRAIN_VERSION)\n", + "DEPLOY_IMAGE = \"gcr.io/cloud-aiplatform/prediction/{}:latest\".format(DEPLOY_VERSION)\n", + "\n", + "print(\"Training:\", TRAIN_IMAGE, TRAIN_GPU, TRAIN_NGPU)\n", + "print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU, DEPLOY_NGPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:training,prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for training and prediction.\n", + "\n", + "- Set the variables `TRAIN_COMPUTE` and `DEPLOY_COMPUTE` to configure the compute resources for the VMs you will use for for training and prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: The following is not supported for training:*\n", + "\n", + " - `standard`: 2 vCPUs\n", + " - `highcpu`: 2, 4 and 8 vCPUs\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_TRAIN_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Train machine type\", TRAIN_COMPUTE)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own custom model and training for CIFAR10." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:custom" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Model Service for `Model` resources.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:custom" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"job\"] = create_job_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_model" + }, + "source": [ + "## Train a model\n", + "\n", + "There are two ways you can train a custom model using a container image:\n", + "\n", + "- **Use a Google Cloud prebuilt container**. If you use a prebuilt container, you will additionally specify a Python package to install into the container image. This Python package contains your code for training a custom model.\n", + "\n", + "- **Use your own custom container image**. If you use your own container, the container needs to contain your code for training a custom model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_specification:prebuilt_container" + }, + "source": [ + "## Prepare your custom job specification\n", + "\n", + "Now that your clients are ready, your first step is to create a Job Specification for your custom training job. The job specification will consist of the following:\n", + "\n", + "- `worker_pool_spec` : The specification of the type of machine(s) you will use for training and how many (single or distributed)\n", + "- `python_package_spec` : The specification of the Python package to be installed with the pre-built container." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "source": [ + "### Prepare your machine specification\n", + "\n", + "Now define the machine specification for your custom training job. This tells Vertex what type of machine instance to provision for the training.\n", + " - `machine_type`: The type of GCP instance to provision -- e.g., n1-standard-8.\n", + " - `accelerator_type`: The type, if any, of hardware accelerator. In this tutorial if you previously set the variable `TRAIN_GPU != None`, you are using a GPU; otherwise you will use a CPU.\n", + " - `accelerator_count`: The number of accelerators." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "outputs": [], + "source": [ + "if TRAIN_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": TRAIN_COMPUTE,\n", + " \"accelerator_type\": TRAIN_GPU,\n", + " \"accelerator_count\": TRAIN_NGPU,\n", + " }\n", + "else:\n", + " machine_spec = {\"machine_type\": TRAIN_COMPUTE, \"accelerator_count\": 0}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "source": [ + "### Prepare your disk specification\n", + "\n", + "(optional) Now define the disk specification for your custom training job. This tells Vertex what type and size of disk to provision in each machine instance for the training.\n", + "\n", + " - `boot_disk_type`: Either SSD or Standard. SSD is faster, and Standard is less expensive. Defaults to SSD.\n", + " - `boot_disk_size_gb`: Size of disk in GB." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "outputs": [], + "source": [ + "DISK_TYPE = \"pd-ssd\" # [ pd-ssd, pd-standard]\n", + "DISK_SIZE = 200 # GB\n", + "\n", + "disk_spec = {\"boot_disk_type\": DISK_TYPE, \"boot_disk_size_gb\": DISK_SIZE}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container" + }, + "source": [ + "### Define the worker pool specification\n", + "\n", + "Next, you define the worker pool specification for your custom training job. The worker pool specification will consist of the following:\n", + "\n", + "- `replica_count`: The number of instances to provision of this machine type.\n", + "- `machine_spec`: The hardware specification.\n", + "- `disk_spec` : (optional) The disk storage specification.\n", + "\n", + "- `python_package`: The Python training package to install on the VM instance(s) and which Python module to invoke, along with command line arguments for the Python module.\n", + "\n", + "Let's dive deeper now into the python package specification:\n", + "\n", + "-`executor_image_spec`: This is the docker image which is configured for your custom training job.\n", + "\n", + "-`package_uris`: This is a list of the locations (URIs) of your python training packages to install on the provisioned instance. The locations need to be in a Cloud Storage bucket. These can be either individual python files or a zip (archive) of an entire package. In the later case, the job service will unzip (unarchive) the contents into the docker image.\n", + "\n", + "-`python_module`: The Python module (script) to invoke for running the custom training job. In this example, you will be invoking `trainer.task.py` -- note that it was not neccessary to append the `.py` suffix.\n", + "\n", + "-`args`: The command line arguments to pass to the corresponding Pythom module. In this example, you will be setting:\n", + " - `\"--model-dir=\" + MODEL_DIR` : The Cloud Storage location where to store the model artifacts. There are two ways to tell the training script where to save the model artifacts:\n", + " - direct: You pass the Cloud Storage location as a command line argument to your training script (set variable `DIRECT = True`), or\n", + " - indirect: The service passes the Cloud Storage location as the environment variable `AIP_MODEL_DIR` to your training script (set variable `DIRECT = False`). In this case, you tell the service the model artifact location in the job specification.\n", + " - `\"--epochs=\" + EPOCHS`: The number of epochs for training.\n", + " - `\"--steps=\" + STEPS`: The number of steps (batches) per epoch.\n", + " - `\"--distribute=\" + TRAIN_STRATEGY\"` : The training distribution strategy to use for single or distributed training.\n", + " - `\"single\"`: single device.\n", + " - `\"mirror\"`: all GPU devices on a single compute instance.\n", + " - `\"multi\"`: all GPU devices on all compute instances." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_job_\" + TIMESTAMP\n", + "MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, JOB_NAME)\n", + "\n", + "if not TRAIN_NGPU or TRAIN_NGPU < 2:\n", + " TRAIN_STRATEGY = \"single\"\n", + "else:\n", + " TRAIN_STRATEGY = \"mirror\"\n", + "\n", + "EPOCHS = 20\n", + "STEPS = 100\n", + "\n", + "DIRECT = True\n", + "if DIRECT:\n", + " CMDARGS = [\n", + " \"--model-dir=\" + MODEL_DIR,\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "else:\n", + " CMDARGS = [\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "\n", + "worker_pool_spec = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": machine_spec,\n", + " \"disk_spec\": disk_spec,\n", + " \"python_package_spec\": {\n", + " \"executor_image_uri\": TRAIN_IMAGE,\n", + " \"package_uris\": [BUCKET_NAME + \"/trainer_cifar10.tar.gz\"],\n", + " \"python_module\": \"trainer.task\",\n", + " \"args\": CMDARGS,\n", + " },\n", + " }\n", + "]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "assemble_custom_job_specification" + }, + "source": [ + "### Assemble a job specification\n", + "\n", + "Now assemble the complete description for the custom job specification:\n", + "\n", + "- `display_name`: The human readable name you assign to this custom job.\n", + "- `job_spec`: The specification for the custom job.\n", + " - `worker_pool_specs`: The specification for the machine VM instances.\n", + " - `base_output_directory`: This tells the service the Cloud Storage location where to save the model artifacts (when variable `DIRECT = False`). The service will then pass the location to the training script as the environment variable `AIP_MODEL_DIR`, and the path will be of the form:\n", + "\n", + " /model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "assemble_custom_job_specification" + }, + "outputs": [], + "source": [ + "if DIRECT:\n", + " job_spec = {\"worker_pool_specs\": worker_pool_spec}\n", + "else:\n", + " job_spec = {\n", + " \"worker_pool_specs\": worker_pool_spec,\n", + " \"base_output_directory\": {\"output_uri_prefix\": MODEL_DIR},\n", + " }\n", + "\n", + "custom_job = {\"display_name\": JOB_NAME, \"job_spec\": job_spec}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "examine_training_package" + }, + "source": [ + "### Examine the training package\n", + "\n", + "#### Package layout\n", + "\n", + "Before you start the training, you will look at how a Python package is assembled for a custom training job. When unarchived, the package contains the following directory/file layout.\n", + "\n", + "- PKG-INFO\n", + "- README.md\n", + "- setup.cfg\n", + "- setup.py\n", + "- trainer\n", + " - \\_\\_init\\_\\_.py\n", + " - task.py\n", + "\n", + "The files `setup.cfg` and `setup.py` are the instructions for installing the package into the operating environment of the Docker image.\n", + "\n", + "The file `trainer/task.py` is the Python script for executing the custom training job. *Note*, when we referred to it in the worker pool specification, we replace the directory slash with a dot (`trainer.task`) and dropped the file suffix (`.py`).\n", + "\n", + "#### Package Assembly\n", + "\n", + "In the following cells, you will assemble the training package." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "examine_training_package" + }, + "outputs": [], + "source": [ + "# Make folder for Python training script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\ntag_build =\\n\\ntag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'tensorflow_datasets==1.3.0',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\nName: CIFAR10 image classification\\n\\nVersion: 0.0.0\\n\\nSummary: Demostration training script\\n\\nHome-page: www.google.com\\n\\nAuthor: Google\\n\\nAuthor-email: aferlitsch@google.com\\n\\nLicense: Public\\n\\nDescription: Demo\\n\\nPlatform: Vertex\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:cifar10" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the training script task.py. We won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n", + "- Loads CIFAR10 dataset from TF Datasets (tfds).\n", + "- Builds a model using TF.Keras model API.\n", + "- Compiles the model (`compile()`).\n", + "- Sets a training distribution strategy according to the argument `args.distribute`.\n", + "- Trains the model (`fit()`) with epochs and steps according to the arguments `args.epochs` and `args.steps`\n", + "- Saves the trained model (`save(args.model_dir)`) to the specified model directory." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:cifar10" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Single, Mirror and Multi-Machine Distributed Training for CIFAR-10\n", + "\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "import argparse\n", + "import os\n", + "import sys\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv(\"AIP_MODEL_DIR\"), type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.01, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=10, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=200, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "print('DEVICES', device_lib.list_local_devices())\n", + "\n", + "# Single Machine, single compute device\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "# Single Machine, multiple compute device\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "# Multiple Machine, multiple compute device\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "# Multi-worker configuration\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "# Preparing dataset\n", + "BUFFER_SIZE = 10000\n", + "BATCH_SIZE = 64\n", + "\n", + "\n", + "def make_datasets_unbatched():\n", + "\n", + " # Scaling CIFAR10 data from (0, 255] to (0., 1.]\n", + " def scale(image, label):\n", + " image = tf.cast(image, tf.float32)\n", + " image /= 255.0\n", + " return image, label\n", + "\n", + "\n", + " datasets, info = tfds.load(name='cifar10',\n", + " with_info=True,\n", + " as_supervised=True)\n", + " return datasets['train'].map(scale).cache().shuffle(BUFFER_SIZE).repeat()\n", + "\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_cnn_model():\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu', input_shape=(32, 32, 3)),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu'),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Flatten(),\n", + " tf.keras.layers.Dense(10, activation='softmax')\n", + " ])\n", + " model.compile(\n", + " loss=tf.keras.losses.sparse_categorical_crossentropy,\n", + " optimizer=tf.keras.optimizers.SGD(learning_rate=args.lr),\n", + " metrics=['accuracy'])\n", + " return model\n", + "\n", + "\n", + "# Train the model\n", + "NUM_WORKERS = strategy.num_replicas_in_sync\n", + "# Here the batch size scales up by number of workers since\n", + "# `tf.data.Dataset.batch` expects the global batch size.\n", + "GLOBAL_BATCH_SIZE = BATCH_SIZE * NUM_WORKERS\n", + "train_dataset = make_datasets_unbatched().batch(GLOBAL_BATCH_SIZE)\n", + "\n", + "with strategy.scope():\n", + " # Creation of dataset, and model building/compiling need to be within\n", + " # `strategy.scope()`.\n", + " model = build_and_compile_cnn_model()\n", + "\n", + "model.fit(x=train_dataset, epochs=args.epochs, steps_per_epoch=args.steps)\n", + "model.save(args.model_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tarball_training_script" + }, + "source": [ + "#### Store training script on your Cloud Storage bucket\n", + "\n", + "Next, you package the training folder into a compressed tar ball, and then store it in your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tarball_training_script" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_cifar10.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job" + }, + "source": [ + "### Train the model\n", + "\n", + "\n", + "Now start the training of your custom training job on Vertex. Use this helper function `create_custom_job`, which takes the following parameter:\n", + "\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "The helper function calls job client service's `create_custom_job` method, with the following parameters:\n", + "\n", + "-`parent`: The Vertex location path to `Dataset`, `Model` and `Endpoint` resources.\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "You will display a handful of the fields returned in `response` object, with the two that are of most interest are:\n", + "\n", + "`response.name`: The Vertex fully qualified identifier assigned to this custom training job. You save this identifier for using in subsequent steps.\n", + "\n", + "`response.state`: The current state of the custom training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job" + }, + "outputs": [], + "source": [ + "def create_custom_job(custom_job):\n", + " response = clients[\"job\"].create_custom_job(parent=PARENT, custom_job=custom_job)\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = create_custom_job(custom_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "job_id:response" + }, + "source": [ + "Now get the unique identifier for the custom job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom job\n", + "job_id = response.name\n", + "# The short numeric ID for the custom job\n", + "job_short_id = job_id.split(\"/\")[-1]\n", + "\n", + "print(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_custom_job" + }, + "source": [ + "### Get information on a custom job\n", + "\n", + "Next, use this helper function `get_custom_job`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "The helper function calls the job client service's`get_custom_job` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "If you recall, you got the Vertex fully qualified identifier for the custom job in the `response.name` field when you called the `create_custom_job` method, and saved the identifier in the variable `job_id`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_custom_job" + }, + "outputs": [], + "source": [ + "def get_custom_job(name, silent=False):\n", + " response = clients[\"job\"].get_custom_job(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = get_custom_job(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete:custom" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, we will need to know the location of the saved model, which the Python script saved in your local Cloud Storage bucket at `MODEL_DIR + '/saved_model.pb'`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete:custom" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_custom_job(job_id, True)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_path_to_deploy = None\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " if not DIRECT:\n", + " MODEL_DIR = MODEL_DIR + \"/model\"\n", + " model_path_to_deploy = MODEL_DIR\n", + " print(\"Training Time:\", response.update_time - response.create_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model_to_deploy:\", model_path_to_deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "load_saved_model" + }, + "source": [ + "## Load the saved model\n", + "\n", + "Your model is stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Now load it from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n", + "\n", + "To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "load_saved_model" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:image" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load the CIFAR10 test (holdout) data from `tf.keras.datasets`, using the method `load_data()`. This will return the dataset as a tuple of two elements. The first element is the training data and the second is the test data. Each element is also a tuple of two elements: the image data, and the corresponding labels.\n", + "\n", + "You don't need the training data, and hence why we loaded it as `(_, _)`.\n", + "\n", + "Before you can run the data through evaluation, you need to preprocess it:\n", + "\n", + "x_test:\n", + "1. Normalize (rescaling) the pixel data by dividing each pixel by 255. This will replace each single byte integer pixel with a 32-bit floating point number between 0 and 1.\n", + "\n", + "y_test:
\n", + "2. The labels are currently scalar (sparse). If you look back at the `compile()` step in the `trainer/task.py` script, you will find that it was compiled for sparse labels. So we don't need to do anything more." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "evaluate_custom_model:image,cifar10" + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "from tensorflow.keras.datasets import cifar10\n", + "\n", + "(_, _), (x_test, y_test) = cifar10.load_data()\n", + "x_test = (x_test / 255.0).astype(np.float32)\n", + "\n", + "print(x_test.shape, y_test.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "perform_evaluation_custom" + }, + "source": [ + "### Perform the model evaluation\n", + "\n", + "Now evaluate how well the model in the custom job did." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "perform_evaluation_custom" + }, + "outputs": [], + "source": [ + "model.evaluate(x_test, y_test)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "how_serving_function_works" + }, + "source": [ + "## Upload the model for serving\n", + "\n", + "Next, you will upload your TF.Keras model from the custom job to Vertex `Model` service, which will create a Vertex `Model` resource for your custom model. During upload, you need to define a serving function to convert data to the format your model expects. If you send encoded data to Vertex, your serving function ensures that the data is decoded on the model server before it is passed as input to your model.\n", + "\n", + "### How does the serving function work\n", + "\n", + "When you send a request to an online prediction server, the request is received by a HTTP server. The HTTP server extracts the prediction request from the HTTP request content body. The extracted prediction request is forwarded to the serving function. For Google pre-built prediction containers, the request content is passed to the serving function as a `tf.string`.\n", + "\n", + "The serving function consists of two parts:\n", + "\n", + "- `preprocessing function`:\n", + " - Converts the input (`tf.string`) to the input shape and data type of the underlying model (dynamic graph).\n", + " - Performs the same preprocessing of the data that was done during training the underlying model -- e.g., normalizing, scaling, etc.\n", + "- `post-processing function`:\n", + " - Converts the model output to format expected by the receiving application -- e.q., compresses the output.\n", + " - Packages the output for the the receiving application -- e.g., add headings, make JSON object, etc.\n", + "\n", + "Both the preprocessing and post-processing functions are converted to static graphs which are fused to the model. The output from the underlying model is passed to the post-processing function. The post-processing function passes the converted/packaged output back to the HTTP server. The HTTP server returns the output as the HTTP response content.\n", + "\n", + "One consideration you need to consider when building serving functions for TF.Keras models is that they run as static graphs. That means, you cannot use TF graph operations that require a dynamic graph. If you do, you will get an error during the compile of the serving function which will indicate that you are using an EagerTensor which is not supported." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_image:xai" + }, + "source": [ + "### Serving function for image data\n", + "\n", + "To pass images to the prediction service, you encode the compressed (e.g., JPEG) image bytes into base 64 -- which makes the content safe from modification while transmitting binary data over the network. Since this deployed model expects input data as raw (uncompressed) bytes, you need to ensure that the base 64 encoded data gets converted back to raw bytes before it is passed as input to the deployed model.\n", + "\n", + "To resolve this, define a serving function (`serving_fn`) and attach it to the model as a preprocessing step. Add a `@tf.function` decorator so the serving function is fused to the underlying model (instead of upstream on a CPU).\n", + "\n", + "When you send a prediction or explanation request, the content of the request is base 64 decoded into a Tensorflow string (`tf.string`), which is passed to the serving function (`serving_fn`). The serving function preprocesses the `tf.string` into raw (uncompressed) numpy bytes (`preprocess_fn`) to match the input requirements of the model:\n", + "- `io.decode_jpeg`- Decompresses the JPG image which is returned as a Tensorflow tensor with three channels (RGB).\n", + "- `image.convert_image_dtype` - Changes integer pixel values to float 32.\n", + "- `image.resize` - Resizes the image to match the input shape for the model.\n", + "- `resized / 255.0` - Rescales (normalization) the pixel data between 0 and 1.\n", + "\n", + "At this point, the data can be passed to the model (`m_call`).\n", + "\n", + "#### XAI Signatures\n", + "\n", + "When the serving function is saved back with the underlying model (`tf.saved_model.save`), you specify the input layer of the serving function as the signature `serving_default`.\n", + "\n", + "For XAI image models, you need to save two additional signatures from the serving function:\n", + "\n", + "- `xai_preprocess`: The preprocessing function in the serving function.\n", + "- `xai_model`: The concrete function for calling the model." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_image:xai" + }, + "outputs": [], + "source": [ + "CONCRETE_INPUT = \"numpy_inputs\"\n", + "\n", + "\n", + "def _preprocess(bytes_input):\n", + " decoded = tf.io.decode_jpeg(bytes_input, channels=3)\n", + " decoded = tf.image.convert_image_dtype(decoded, tf.float32)\n", + " resized = tf.image.resize(decoded, size=(32, 32))\n", + " rescale = tf.cast(resized / 255.0, tf.float32)\n", + " return rescale\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def preprocess_fn(bytes_inputs):\n", + " decoded_images = tf.map_fn(\n", + " _preprocess, bytes_inputs, dtype=tf.float32, back_prop=False\n", + " )\n", + " return {\n", + " CONCRETE_INPUT: decoded_images\n", + " } # User needs to make sure the key matches model's input\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def serving_fn(bytes_inputs):\n", + " images = preprocess_fn(bytes_inputs)\n", + " prob = m_call(**images)\n", + " return prob\n", + "\n", + "\n", + "m_call = tf.function(model.call).get_concrete_function(\n", + " [tf.TensorSpec(shape=[None, 32, 32, 3], dtype=tf.float32, name=CONCRETE_INPUT)]\n", + ")\n", + "\n", + "tf.saved_model.save(\n", + " model,\n", + " model_path_to_deploy,\n", + " signatures={\n", + " \"serving_default\": serving_fn,\n", + " # Required for XAI\n", + " \"xai_preprocess\": preprocess_fn,\n", + " \"xai_model\": m_call,\n", + " },\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_signature:xai" + }, + "source": [ + "## Get the serving function signature\n", + "\n", + "You can get the signatures of your model's input and output layers by reloading the model into memory, and querying it for the signatures corresponding to each layer.\n", + "\n", + "When making a prediction request, you need to route the request to the serving function instead of the model, so you need to know the input layer name of the serving function -- which you will use later when you make a prediction request.\n", + "\n", + "You also need to know the name of the serving function's input and output layer for constructing the explanation metadata -- which is discussed subsequently." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_signature:xai" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_path_to_deploy)\n", + "\n", + "serving_input = list(\n", + " loaded.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "print(\"Serving function input:\", serving_input)\n", + "serving_output = list(loaded.signatures[\"serving_default\"].structured_outputs.keys())[0]\n", + "print(\"Serving function output:\", serving_output)\n", + "\n", + "input_name = model.input.name\n", + "print(\"Model input name:\", input_name)\n", + "output_name = model.output.name\n", + "print(\"Model output name:\", output_name)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "explanation_spec" + }, + "source": [ + "### Explanation Specification\n", + "\n", + "To get explanations when doing a prediction, you must enable the explanation capability and set corresponding settings when you upload your custom model to an Vertex `Model` resource. These settings are referred to as the explanation metadata, which consists of:\n", + "\n", + "- `parameters`: This is the specification for the explainability algorithm to use for explanations on your model. You can choose between:\n", + " - Shapley - *Note*, not recommended for image data -- can be very long running\n", + " - XRAI\n", + " - Integrated Gradients\n", + "- `metadata`: This is the specification for how the algoithm is applied on your custom model.\n", + "\n", + "#### Explanation Parameters\n", + "\n", + "Let's first dive deeper into the settings for the explainability algorithm.\n", + "\n", + "#### Shapley\n", + "\n", + "Assigns credit for the outcome to each feature, and considers different permutations of the features. This method provides a sampling approximation of exact Shapley values.\n", + "\n", + "Use Cases:\n", + " - Classification and regression on tabular data.\n", + "\n", + "Parameters:\n", + "\n", + "- `path_count`: This is the number of paths over the features that will be processed by the algorithm. An exact approximation of the Shapley values requires M! paths, where M is the number of features. For the CIFAR10 dataset, this would be 784 (28*28).\n", + "\n", + "For any non-trival number of features, this is too compute expensive. You can reduce the number of paths over the features to M * `path_count`.\n", + "\n", + "#### Integrated Gradients\n", + "\n", + "A gradients-based method to efficiently compute feature attributions with the same axiomatic properties as the Shapley value.\n", + "\n", + "Use Cases:\n", + " - Classification and regression on tabular data.\n", + " - Classification on image data.\n", + "\n", + "Parameters:\n", + "\n", + "- `step_count`: This is the number of steps to approximate the remaining sum. The more steps, the more accurate the integral approximation. The general rule of thumb is 50 steps, but as you increase so does the compute time.\n", + "\n", + "#### XRAI\n", + "\n", + "Based on the integrated gradients method, XRAI assesses overlapping regions of the image to create a saliency map, which highlights relevant regions of the image rather than pixels.\n", + "\n", + "Use Cases:\n", + "\n", + " - Classification on image data.\n", + "\n", + "Parameters:\n", + "\n", + "- `step_count`: This is the number of steps to approximate the remaining sum. The more steps, the more accurate the integral approximation. The general rule of thumb is 50 steps, but as you increase so does the compute time.\n", + "\n", + "In the next code cell, set the variable `XAI` to which explainabilty algorithm you will use on your custom model." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "explanation_spec" + }, + "outputs": [], + "source": [ + "XAI = \"ig\" # [ shapley, ig, xrai ]\n", + "\n", + "if XAI == \"shapley\":\n", + " PARAMETERS = {\"sampled_shapley_attribution\": {\"path_count\": 10}}\n", + "elif XAI == \"ig\":\n", + " PARAMETERS = {\"integrated_gradients_attribution\": {\"step_count\": 50}}\n", + "elif XAI == \"xrai\":\n", + " PARAMETERS = {\"xrai_attribution\": {\"step_count\": 50}}\n", + "\n", + "parameters = aip.ExplanationParameters(PARAMETERS)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "explanation_metadata:image" + }, + "source": [ + "#### Explanation Metadata\n", + "\n", + "Let's first dive deeper into the explanation metadata, which consists of:\n", + "\n", + "- `outputs`: A scalar value in the output to attribute -- what to explain. For example, in a probability output \\[0.1, 0.2, 0.7\\] for classification, one wants an explanation for 0.7. Consider the following formulae, where the output is `y` and that is what we want to explain.\n", + "\n", + " y = f(x)\n", + "\n", + "Consider the following formulae, where the outputs are `y` and `z`. Since we can only do attribution for one scalar value, we have to pick whether we want to explain the output `y` or `z`. Assume in this example the model is object detection and y and z are the bounding box and the object classification. You would want to pick which of the two outputs to explain.\n", + "\n", + " y, z = f(x)\n", + "\n", + "The dictionary format for `outputs` is:\n", + "\n", + " { \"outputs\": { \"[your_display_name]\":\n", + " \"output_tensor_name\": [layer]\n", + " }\n", + " }\n", + "\n", + "
\n", + " - [your_display_name]: A human readable name you assign to the output to explain. A common example is \"probability\".
\n", + " - \"output_tensor_name\": The key/value field to identify the output layer to explain.
\n", + " - [layer]: The output layer to explain. In a single task model, like a tabular regressor, it is the last (topmost) layer in the model.\n", + "
\n", + "\n", + "- `inputs`: The features for attribution -- how they contributed to the output. Consider the following formulae, where `a` and `b` are the features. We have to pick which features to explain how the contributed. Assume that this model is deployed for A/B testing, where `a` are the data_items for the prediction and `b` identifies whether the model instance is A or B. You would want to pick `a` (or some subset of) for the features, and not `b` since it does not contribute to the prediction.\n", + "\n", + " y = f(a,b)\n", + "\n", + "The minimum dictionary format for `inputs` is:\n", + "\n", + " { \"inputs\": { \"[your_display_name]\":\n", + " \"input_tensor_name\": [layer]\n", + " }\n", + " }\n", + "\n", + "
\n", + " - [your_display_name]: A human readable name you assign to the input to explain. A common example is \"features\".
\n", + " - \"input_tensor_name\": The key/value field to identify the input layer for the feature attribution.
\n", + " - [layer]: The input layer for feature attribution. In a single input tensor model, it is the first (bottom-most) layer in the model.\n", + "
\n", + "\n", + "Since the inputs to the model are tabular, you can specify the following two additional fields as reporting/visualization aids:\n", + "\n", + "
\n", + " - \"modality\": \"image\": Indicates the field values are image data.\n", + "
\n", + "\n", + "Since the inputs to the model are images, you can specify the following additional fields as reporting/visualization aids:\n", + "\n", + "
\n", + " - \"modality\": \"image\": Indicates the field values are image data.\n", + "
" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "explanation_metadata:image" + }, + "outputs": [], + "source": [ + "random_baseline = np.random.rand(32, 32, 3)\n", + "input_baselines = [{\"number_vaue\": x} for x in random_baseline]\n", + "\n", + "INPUT_METADATA = {\"input_tensor_name\": CONCRETE_INPUT, \"modality\": \"image\"}\n", + "\n", + "OUTPUT_METADATA = {\"output_tensor_name\": serving_output}\n", + "\n", + "input_metadata = aip.ExplanationMetadata.InputMetadata(INPUT_METADATA)\n", + "output_metadata = aip.ExplanationMetadata.OutputMetadata(OUTPUT_METADATA)\n", + "\n", + "metadata = aip.ExplanationMetadata(\n", + " inputs={\"image\": input_metadata}, outputs={\"class\": output_metadata}\n", + ")\n", + "\n", + "explanation_spec = aip.ExplanationSpec(metadata=metadata, parameters=parameters)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "upload_the_model:explanation" + }, + "source": [ + "### Upload the model\n", + "\n", + "Use this helper function `upload_model` to upload your model, stored in SavedModel format, up to the `Model` service, which will instantiate a Vertex `Model` resource instance for your model. Once you've done that, you can use the `Model` resource instance in the same way as any other Vertex `Model` resource instance, such as deploying to an `Endpoint` resource for serving predictions.\n", + "\n", + "The helper function takes the following parameters:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` service.\n", + "- `image_uri`: The container image for the model deployment.\n", + "- `model_uri`: The Cloud Storage path to our SavedModel artificat. For this tutorial, this is the Cloud Storage location where the `trainer/task.py` saved the model artifacts, which we specified in the variable `MODEL_DIR`.\n", + "\n", + "The helper function calls the `Model` client service's method `upload_model`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for `Dataset`, `Model` and `Endpoint` resources.\n", + "- `model`: The specification for the Vertex `Model` resource instance.\n", + "\n", + "Let's now dive deeper into the Vertex model specification `model`. This is a dictionary object that consists of the following fields:\n", + "\n", + "- `display_name`: A human readable name for the `Model` resource.\n", + "- `metadata_schema_uri`: Since your model was built without an Vertex `Dataset` resource, you will leave this blank (`''`).\n", + "- `artificat_uri`: The Cloud Storage path where the model is stored in SavedModel format.\n", + "- `container_spec`: This is the specification for the Docker container that will be installed on the `Endpoint` resource, from which the `Model` resource will serve predictions. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + "- `explanation_spec`: This is the specification for enabling explainability for your model.\n", + "\n", + "Uploading a model into a Vertex Model resource returns a long running operation, since it may take a few moments. You call response.result(), which is a synchronous call and will return when the Vertex Model resource is ready.\n", + "\n", + "The helper function returns the Vertex fully qualified identifier for the corresponding Vertex Model instance upload_model_response.model. You will save the identifier for subsequent steps in the variable model_to_deploy_id." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "upload_the_model:explanation" + }, + "outputs": [], + "source": [ + "IMAGE_URI = DEPLOY_IMAGE\n", + "\n", + "\n", + "def upload_model(display_name, image_uri, model_uri):\n", + "\n", + " model = aip.Model(\n", + " display_name=display_name,\n", + " artifact_uri=model_uri,\n", + " metadata_schema_uri=\"\",\n", + " explanation_spec=explanation_spec,\n", + " container_spec={\"image_uri\": image_uri},\n", + " )\n", + "\n", + " response = clients[\"model\"].upload_model(parent=PARENT, model=model)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " upload_model_response = response.result(timeout=180)\n", + " print(\"upload_model_response\")\n", + " print(\" model:\", upload_model_response.model)\n", + " return upload_model_response.model\n", + "\n", + "\n", + "model_to_deploy_id = upload_model(\n", + " \"cifar10-\" + TIMESTAMP, IMAGE_URI, model_path_to_deploy\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_model" + }, + "source": [ + "### Get `Model` resource information\n", + "\n", + "Now let's get the model information for just your model. Use this helper function `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource.\n", + "\n", + "This helper function calls the Vertex `Model` client service's method `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_model" + }, + "outputs": [], + "source": [ + "def get_model(name):\n", + " response = clients[\"model\"].get_model(name=name)\n", + " print(response)\n", + "\n", + "\n", + "get_model(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:custom" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex custom `Model` resource. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"cifar10_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:dedicated,v1beta1" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the model to the endpoint you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified identifier of the `Model` resource to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to to.\n", + "- `deployed_model`: The requirements for deploying the model.\n", + "- `traffic_split`: Percent of traffic at endpoint that goes to this model, which is specified as a dictioney of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then specify as, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + " { \"0\": percent, model_id: percent, ... }\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified identifier of the (upload) `Model` resource to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `dedicated_resources`: This refers to how many compute instances (replicas) that are scaled for serving prediction requests.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `min_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "- `enable_container_logging`: This enables logging of container events, such as execution failures (default is container logging is disabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:dedicated,v1beta1" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"cifar10_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " \"machine_spec\": machine_spec,\n", + " },\n", + " \"enable_container_logging\": False,\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_item:test" + }, + "source": [ + "### Get test item\n", + "\n", + "You will use an example out of the test (holdout) portion of the dataset as a test item." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:test" + }, + "outputs": [], + "source": [ + "test_image = x_test[0]\n", + "test_label = y_test[0]\n", + "print(test_image.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "prepare_test_item:test,image" + }, + "source": [ + "### Prepare the request content\n", + "You are going to send the CIFAR10 image as compressed JPG image, instead of the raw uncompressed bytes:\n", + "\n", + "- `cv2.imwrite`: Use openCV to write the uncompressed image to disk as a compressed JPEG image.\n", + " - Denormalize the image data from \\[0,1) range back to [0,255).\n", + " - Convert the 32-bit floating point values to 8-bit unsigned integers.\n", + "- `tf.io.read_file`: Read the compressed JPG images back into memory as raw bytes.\n", + "- `base64.b64encode`: Encode the raw bytes into a base 64 encoded string." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "prepare_test_item:test,image" + }, + "outputs": [], + "source": [ + "import base64\n", + "\n", + "import cv2\n", + "\n", + "cv2.imwrite(\"tmp.jpg\", (test_image * 255).astype(np.uint8))\n", + "\n", + "bytes = tf.io.read_file(\"tmp.jpg\")\n", + "b64str = base64.b64encode(bytes.numpy()).decode(\"utf-8\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "send_explain_request:image" + }, + "source": [ + "### Send the prediction with explanation request\n", + "\n", + "Ok, now you have a test image. Use this helper function `explain_image`, which takes the parameters:\n", + "\n", + "- `image`: A list of test image data as a numpy array.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `parameters_dict`: Additional parameters for serving.\n", + "- `deployed_model_id`: The Vertex fully qualified identifier for the deployed model, when more than one model is deployed at the endpoint. Otherwise, if only one model deployed, can be set to `None`.\n", + "\n", + "This function uses the prediction client service and calls the `explain` method with the parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `instances`: A list of instances (encoded images) to predict and explain.\n", + "- `parameters`: Additional parameters for serving.\n", + "- `deployed_model_id`: The Vertex fully qualified identifier for the deployed model, when more than one model is deployed at the endpoint. Otherwise, if only one model deployed, can be set to `None`.\n", + "\n", + "To pass the image data to the prediction service, in the previous step you encoded the bytes into base 64 -- which makes the content safe from modification when transmitting binary data over the network. You need to tell the serving binary where your model is deployed to, that the content has been base 64 encoded, so it will decode it on the other end in the serving binary.\n", + "\n", + "Each instance in the prediction request is a dictionary entry of the form:\n", + "\n", + " {input_name: {'b64': content}}\n", + "\n", + "- `input_name`: the name of the input layer of the underlying model.\n", + "- `'b64'`: A key that indicates the content is base 64 encoded.\n", + "- `content`: The compressed JPG image bytes as a base 64 encoded string.\n", + "\n", + "Since the `predict()` service can take multiple images (instances), you will send your single image as a list of one image. As a final step, you package the instances list into Google's protobuf format -- which is what we pass to the `explain()` service.\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding image in the request. You will see in the output for each prediction:\n", + "\n", + "- `deployed_model_id` -- The Vertex fully qualified identifer for the model that did the prediction/explanation.\n", + "- `predictions` -- Confidence level for the prediction (`predictions`), between 0 and 1, for each of the ten classes.\n", + "- `explanations` -- How each feature contributed to the prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "send_explain_request:image" + }, + "outputs": [], + "source": [ + "def explain_image(image, endpoint, parameters_dict, deployed_model_id):\n", + " parameters = json_format.ParseDict(parameters_dict, Value())\n", + "\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [{serving_input: {\"b64\": image}}]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].explain(\n", + " endpoint=endpoint,\n", + " instances=instances,\n", + " parameters=parameters,\n", + " deployed_model_id=deployed_model_id,\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", prediction)\n", + "\n", + " explanations = response.explanations\n", + " print(\"explanations\")\n", + " for explanation in explanations:\n", + " print(\" explanation:\", explanation)\n", + "\n", + " return response\n", + "\n", + "\n", + "response = explain_image(b64str, endpoint_id, None, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "understanding_explanations:cifar10" + }, + "source": [ + "### Understanding the explanations response\n", + "\n", + "Preview the images and their predicted classes without the explanations. Why did the model predict these classes?" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "understanding_explanations:cifar10" + }, + "outputs": [], + "source": [ + "from io import BytesIO\n", + "\n", + "import matplotlib.image as mpimg\n", + "import matplotlib.pyplot as plt\n", + "\n", + "CLASSES = [\n", + " \"airplane\",\n", + " \"automobile\",\n", + " \"bird\",\n", + " \"cat\",\n", + " \"deer\",\n", + " \"dog\",\n", + " \"frog\",\n", + " \"horse\",\n", + " \"ship\",\n", + " \"truck\",\n", + "]\n", + "\n", + "# Note: change the `ig_response` variable below if you didn't deploy an IG model\n", + "for prediction in response.predictions:\n", + " label_index = np.argmax(prediction)\n", + " class_name = CLASSES[label_index]\n", + " confidence_score = prediction[label_index]\n", + " print(\n", + " \"Predicted class: \"\n", + " + class_name\n", + " + \"\\n\"\n", + " + \"Confidence score: \"\n", + " + str(confidence_score)\n", + " )\n", + "\n", + " image = base64.b64decode(b64str)\n", + " image = BytesIO(image)\n", + " img = mpimg.imread(image, format=\"JPG\")\n", + "\n", + " plt.imshow(img, interpolation=\"nearest\")\n", + " plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "visualize_image_explanations" + }, + "source": [ + "### Visualize the images with AI Explanations\n", + "\n", + "The images returned show the explanations for only the top class predicted by the model. This means that if one of the model's predictions is incorrect, the pixels you see highlighted are for the *incorrect class*. For example, if the model predicted \"airplane\" when it should have predicted \"cat\", you can see explanations for why the model classified this image as an airplane.\n", + "\n", + "If you deployed an Integrated Gradients model, you can visualize its feature attributions. Currently, the highlighted pixels returned from AI Explanations show the top 60% of pixels that contributed to the model's prediction. The pixels you see after running the cell below show the pixels that most signaled the model's prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "visualize_image_explanations" + }, + "outputs": [], + "source": [ + "import io\n", + "\n", + "for explanation in response.explanations:\n", + " attributions = dict(explanation.attributions[0].feature_attributions)\n", + " label_index = explanation.attributions[0].output_index[0]\n", + " class_name = CLASSES[label_index]\n", + " b64str = attributions[\"image\"][\"b64_jpeg\"]\n", + " image = base64.b64decode(b64str)\n", + " image = io.BytesIO(image)\n", + " img = mpimg.imread(image, format=\"JPG\")\n", + "\n", + " plt.imshow(img, interpolation=\"nearest\")\n", + " plt.show()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_custom_image_classification_online_explain.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_custom_image_classification_online_exported_ds.ipynb b/notebooks/community/gapic/custom/showcase_custom_image_classification_online_exported_ds.ipynb new file mode 100644 index 000000000..1b20bf211 --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_custom_image_classification_online_exported_ds.ipynb @@ -0,0 +1,2795 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Custom training image classification model for online prediction using exported dataset\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom,exported_ds" + }, + "source": [ + "## Overview\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to train and deploy a custom image classification model for online prediction, using an exported `Dataset` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:flowers,icn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Flowers dataset](https://www.tensorflow.org/datasets/catalog/tf_flowers) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts the type of flower an image is from a class of five flowers: daisy, dandelion, rose, sunflower, or tulip." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,exported_ds,online_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you learn how to create a custom model using an exported `Dataset` resource from a Python script in a Docker container using the Vertex client library, and then do a prediction on the deployed model. You can alternatively create models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Export the `Dataset` resource's manifest.\n", + "- Create a Vertex custom job for training a model.\n", + "- Import the exported dataset manifest.\n", + "- Train the model.\n", + "- Retrieve and load the model artifacts.\n", + "- View the model evaluation.\n", + "- Upload the model as a Vertex `Model` resource.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "labeling_constants:icn" + }, + "source": [ + "#### Labeling constants\n", + "\n", + "Set constants unique to `Dataset` resource labeling" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "labeling_constants:icn" + }, + "outputs": [], + "source": [ + "# Image Dataset type\n", + "DATA_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/metadata/image_1.0.0.yaml\"\n", + "# Image Labeling type\n", + "LABEL_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/ioformat/image_classification_single_label_io_format_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for training and prediction.\n", + "\n", + "Set the variables `TRAIN_GPU/TRAIN_NGPU` and `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU.\n", + "\n", + "*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3 -- which is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n", + " TRAIN_GPU, TRAIN_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n", + " )\n", + "else:\n", + " TRAIN_GPU, TRAIN_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (None, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:training,prediction" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for training and prediction\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-3:latest`\n", + " - TensorFlow 2.4\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-4:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/training/xgboost-cpu.1-1`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/training/scikit-learn-cpu.0-23:latest`\n", + " - Pytorch\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-5:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-6:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-7:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for training](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers).\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-3:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-90:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-82:latest`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-23:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-22:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-20:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for prediction](https://cloud.google.com/vertex-ai/docs/predictions/pre-built-containers)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf2-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf2-cpu.{}\".format(TF)\n", + "else:\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "TRAIN_IMAGE = \"gcr.io/cloud-aiplatform/training/{}:latest\".format(TRAIN_VERSION)\n", + "DEPLOY_IMAGE = \"gcr.io/cloud-aiplatform/prediction/{}:latest\".format(DEPLOY_VERSION)\n", + "\n", + "print(\"Training:\", TRAIN_IMAGE, TRAIN_GPU, TRAIN_NGPU)\n", + "print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU, DEPLOY_NGPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:training,prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for training and prediction.\n", + "\n", + "- Set the variables `TRAIN_COMPUTE` and `DEPLOY_COMPUTE` to configure the compute resources for the VMs you will use for for training and prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: The following is not supported for training:*\n", + "\n", + " - `standard`: 2 vCPUs\n", + " - `highcpu`: 2, 4 and 8 vCPUs\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_TRAIN_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Train machine type\", TRAIN_COMPUTE)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own custom model and training for Flowers." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:custom,exported_ds" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:custom,exported_ds" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"job\"] = create_job_client()\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "\n", + "def create_dataset(name, schema, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " dataset = aip.Dataset(\n", + " display_name=name, metadata_schema_uri=schema, labels=labels\n", + " )\n", + "\n", + " operation = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"flowers-\" + TIMESTAMP, DATA_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:image,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for images has some requirements for your data:\n", + "\n", + "- Images must be stored in a Cloud Storage bucket.\n", + "- Each image file must be in an image format (PNG, JPEG, BMP, ...).\n", + "- There must be an index file stored in your Cloud Storage bucket that contains the path and label for each image.\n", + "- The index file must be either CSV or JSONL." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:icn,u_dataset,csv" + }, + "source": [ + "#### CSV\n", + "\n", + "For image classification, the CSV index file has the requirements:\n", + "\n", + "- No heading.\n", + "- First column is the Cloud Storage path to the image.\n", + "- Second column is the label." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:flowers,csv,icn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = (\n", + " \"gs://cloud-samples-data/vision/automl_classification/flowers/all_data_v2.csv\"\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:csv" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Flowers dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:csv" + }, + "outputs": [], + "source": [ + "if \"IMPORT_FILES\" in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_data" + }, + "source": [ + "### Import data\n", + "\n", + "Now, import the data into your Vertex Dataset resource. Use this helper function `import_data` to import the data. The function does the following:\n", + "\n", + "- Uses the `Dataset` client.\n", + "- Calls the client method `import_data`, with the following parameters:\n", + " - `name`: The human readable name you give to the `Dataset` resource (e.g., flowers).\n", + " - `import_configs`: The import configuration.\n", + "\n", + "- `import_configs`: A Python list containing a dictionary, with the key/value entries:\n", + " - `gcs_sources`: A list of URIs to the paths of the one or more index files.\n", + " - `import_schema_uri`: The schema identifying the labeling type.\n", + "\n", + "The `import_data()` method returns a long running `operation` object. This will take a few minutes to complete. If you are in a live tutorial, this would be a good time to ask questions, or take a personal break." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_data" + }, + "outputs": [], + "source": [ + "def import_data(dataset, gcs_sources, schema):\n", + " config = [{\"gcs_source\": {\"uris\": gcs_sources}, \"import_schema_uri\": schema}]\n", + " print(\"dataset:\", dataset_id)\n", + " start_time = time.time()\n", + " try:\n", + " operation = clients[\"dataset\"].import_data(\n", + " name=dataset_id, import_configs=config\n", + " )\n", + " print(\"Long running operation:\", operation.operation.name)\n", + "\n", + " result = operation.result()\n", + " print(\"result:\", result)\n", + " print(\"time:\", int(time.time() - start_time), \"secs\")\n", + " print(\"error:\", operation.exception())\n", + " print(\"meta :\", operation.metadata)\n", + " print(\n", + " \"after: running:\",\n", + " operation.running(),\n", + " \"done:\",\n", + " operation.done(),\n", + " \"cancelled:\",\n", + " operation.cancelled(),\n", + " )\n", + "\n", + " return operation\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "import_data(dataset_id, [IMPORT_FILE], LABEL_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "export_dataset" + }, + "source": [ + "### Export dataset index\n", + "\n", + "Next, you will export the dataset index to a JSONL file which will then be used by your custom training job to get the data and corresponding labels for training your Flowers model. Use this helper function `export_data` to export the dataset index. The function does the following:\n", + "\n", + "- Uses the dataset client.\n", + "- Calls the client method `export_data`, with the following parameters:\n", + " - `name`: The human readable name you give to the dataset (e.g., flowers).\n", + " - `export_config`: The export configuration.\n", + "- `export_config` A python list containing a dictionary, with the key/value entries:\n", + " - `gcs_destination`: The Cloud Storage bucket to write the JSONL dataset index file to.\n", + "\n", + "The `export_data()` method returns a long running `operation` object. This will take a few minutes to complete. The helper function will return the long running operation and the result of the operation when the export has completed." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "export_dataset" + }, + "outputs": [], + "source": [ + "EXPORT_FILE = BUCKET_NAME + \"/export\"\n", + "\n", + "\n", + "def export_data(dataset_id, gcs_dest):\n", + " config = {\"gcs_destination\": {\"output_uri_prefix\": gcs_dest}}\n", + " start_time = time.time()\n", + " try:\n", + " operation = clients[\"dataset\"].export_data(\n", + " name=dataset_id, export_config=config\n", + " )\n", + " print(\"Long running operation:\", operation.operation.name)\n", + "\n", + " result = operation.result()\n", + " print(\"result:\", result)\n", + " print(\"time:\", int(time.time() - start_time), \"secs\")\n", + " print(\"error:\", operation.exception())\n", + " print(\"meta :\", operation.metadata)\n", + " print(\n", + " \"after: running:\",\n", + " operation.running(),\n", + " \"done:\",\n", + " operation.done(),\n", + " \"cancelled:\",\n", + " operation.cancelled(),\n", + " )\n", + "\n", + " return operation, result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None, None\n", + "\n", + "\n", + "_, result = export_data(dataset_id, EXPORT_FILE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "export_dataset_quick_peak:icn" + }, + "source": [ + "#### Quick peak at your exported dataset index file\n", + "\n", + "Let's now take a quick peak at the contents of the exported dataset index file. When the `export_data()` completed, the response object was obtained from the `result()` method of the long running operation. The response object contains the property:\n", + "\n", + "- `exported_files`: A list of the paths to the exported dataset index files, which in this case will be one file.\n", + "\n", + "You will get the path to the exported dataset index file (`result.exported_files[0]`) and then display the first ten JSON objects in the file -- i.e., data items.\n", + "\n", + "The JSONL format for each data item is:\n", + "\n", + " { \"imageGcsUri\": path_to_the_image, \"classificationAnnotation\": { \"displayName\": label } }" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "export_dataset_quick_peak:icn" + }, + "outputs": [], + "source": [ + "jsonl_index = result.exported_files[0]\n", + "\n", + "! gsutil cat $jsonl_index | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "export_dataset_read_file:icn" + }, + "source": [ + "#### Reading the index file\n", + "\n", + "You will need to add code to your custom training Python script to read the exported dataset index, so that you can generate training batches for custom training your model.\n", + "\n", + "Below is an example of how you might each the exported dataset index file:\n", + "\n", + "1. Use Tensorflow's Cloud Storage file methods to open the file (`tf.io.gfile.GFile()`) and read all the lines (`f.readlines()`), where each line is a data item represented as a JSONL object.\n", + "\n", + "2. For each line in the file, convert the line to a JSON object (`json.loads()`).\n", + "\n", + "3. Extract the path to the image (`['imageGcsUri']`) and label (`['classificationAnnotation']['displayName']`)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "export_dataset_read_file:icn" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "import tensorflow as tf\n", + "\n", + "with tf.io.gfile.GFile(jsonl_index, \"r\") as f:\n", + " export_data_items = f.readlines()\n", + "\n", + "for _ in range(10):\n", + " j = json.loads(export_data_items[_])\n", + " print(j[\"imageGcsUri\"], j[\"classificationAnnotation\"][\"displayName\"])" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_tfrecord:image" + }, + "source": [ + "#### Create TFRecords\n", + "\n", + "Next, we needs to create a feeder mechanism to feed data to the model you will train from the dataset index file. There are lots of choices for how to construct a feeder. We will cover a two options here, both using TFRecords:\n", + "\n", + " 1. Storing the image data as raw uncompressed image data (1 byte per pixel).\n", + " 2. Storing the image data as preprocessed data -- machine learning ready -- (4 bytes per pixel).\n", + "\n", + "These two methods demonstrate a trade-off between disk storage and compute time. In both cases, we do a prepass over the image data to cache the data into a form that will accelerate the training time for the model. But, by caching we are both using disk space and increasing I/O traffic from the disk to the compute device -- e.g., CPU, GPU, TPU.\n", + "\n", + "In the raw uncompressed format, you are minimizing the size on disk and I/O traffic for the cache data, but have the overhead that on each epoch, the preprocessing of the image data has to be repeated. In the preprocessed format, you are minimizing the compute time by preprocessing once and caching the preprocessed data -- i.e., machine learning ready. The amount of data on disk will be four times the size when training as Float32, and you are increasing by the same amount disk space and I/O traffic from the disk to the compute engine.\n", + "\n", + "The helper functions `TFExampleImageUncompressed` and `TFExampleImagePreprocessed` both take the parameters:\n", + "\n", + "- `path`: The Cloud Storage path to the image file.\n", + "- `label`: The corresponding label for the image file.\n", + "- `shape`: The (H,W) input shape to resize the image. If `None`, no resizing occurs.\n", + "\n", + "The helper function `TFExampleImagePreprocessed` has an additional parameter:\n", + "\n", + "- `dtype`: The floating point representation after the pixel data has been normalized. By default, it is set to 32-bit float (np.float32). If you are using NVIDIA GPUs or TPUs you can alternatively train in 16-bit float, by setting `dtype = np.float16`. There are two benefits to training with 16-bit float, when it does not effect the accuracy or number of epochs:\n", + "\n", + " 1. Each matrix multiply operation is 4 times faster than the 32-bit equivalent -- albeit the model weights need to be stored as 16-bit as well.\n", + " 2. The disk space and I/O bandwidth is reduced by 1/2.\n", + "\n", + "Let's look at bit deeper into the functions for creating `TFRecord` training data. First, `TFRecord` is a serialized binary encoding of the training data. As an encoding, one needs to specify a schema for how the fields are encoded, which is then used later to decode during when feeding training data to your model.\n", + "\n", + "The schema is defined as an instance of `tf.train.Example` per data item in the training data. Each instance of `tf.train.Example` consists of a sequence fields, each defined as a key/value pair. In our helper function, the key entries are:\n", + "\n", + "- `image`: The encoded raw image data.\n", + "- `label`: The label assigned to the image.\n", + "- `shape`: The shape of the image when decoded.\n", + "\n", + "The value for each key/value pair is an instance of `tf.train.Feature`, where:\n", + "\n", + "- `bytes_list`: the data to encode is a byte string.\n", + "- `int64_list`: the data to encode is an array of one or more integer values.\n", + "- `float_list`: the data to encode is an array of one or more floating point values." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_tfrecord:image" + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "\n", + "\n", + "def TFExampleImageUncompressed(path, label, shape=None):\n", + " \"\"\" The uncompressed version of the image \"\"\"\n", + "\n", + " # read in (and uncompress) the image\n", + " with tf.io.gfile.GFile(path, \"rb\") as f:\n", + " data = f.read()\n", + " image = tf.io.decode_image(data)\n", + "\n", + " if shape:\n", + " image = tf.image.resize(image, shape)\n", + " image = image.numpy()\n", + " shape = image.shape\n", + "\n", + " # make the record\n", + " return tf.train.Example(\n", + " features=tf.train.Features(\n", + " feature={\n", + " \"image\": tf.train.Feature(\n", + " bytes_list=tf.train.BytesList(value=[image.tostring()])\n", + " ),\n", + " \"label\": tf.train.Feature(int64_list=tf.train.Int64List(value=[label])),\n", + " \"shape\": tf.train.Feature(\n", + " int64_list=tf.train.Int64List(value=[shape[0], shape[1], shape[2]])\n", + " ),\n", + " }\n", + " )\n", + " )\n", + "\n", + "\n", + "def TFExampleImagePreprocessed(path, label, shape=None, dtype=np.float32):\n", + " \"\"\" The normalized version of the image \"\"\"\n", + "\n", + " # read in (uncompress) the image and normalize the pixel data\n", + " image = (cv2.imread(path) / 255.0).astype(dtype)\n", + "\n", + " if shape:\n", + " image = tf.image.resize(image, shape)\n", + " image = image.numpy()\n", + " shape = image.shape\n", + "\n", + " # make the record\n", + " return tf.train.Example(\n", + " features=tf.train.Features(\n", + " feature={\n", + " \"image\": tf.train.Feature(\n", + " bytes_list=tf.train.BytesList(value=[image.tostring()])\n", + " ),\n", + " \"label\": tf.train.Feature(int64_list=tf.train.Int64List(value=[label])),\n", + " \"shape\": tf.train.Feature(\n", + " int64_list=tf.train.Int64List(value=[shape[0], shape[1], shape[2]])\n", + " ),\n", + " }\n", + " )\n", + " )" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "write_tfrecord:image" + }, + "source": [ + "#### Write training data to TFRecord file\n", + "\n", + "Next, you will create a single TFRecord for all the training data specified in the exported dataset index:\n", + "\n", + "- Specify the cache method by setting the variable `CACHE` to either `TFExampleImageUncompressed` or `TFExampleImagePreprocessed`.\n", + "- Convert class names from the dataset to integer labels, using `cls2label`.\n", + "- Read in the data item list from the exported dataset index file -- `tf.io.gfile.GFile(jsonl_index, 'r')`.\n", + "- Set the Cloud Storage location to store the cached TFRecord file -- `GCS_TFRECORD_URI`.\n", + "- Generate the cached data using `tf.io.TFRecordWriter(gcs_tfrecord_uri)` for each data item in the exported dataset index.\n", + " - Extract the Cloud Storage path and class name - `json.loads(data_item)`\n", + " - Convert class name to integer label - `label = cls2label[cls]`\n", + " - Encode the data item - `example = CACHE(image, label)`\n", + " - Write the encoded data item to the TFRecord file - `writer.write(example.SerializeToString())`\n", + "\n", + "This may take about 20 minutes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "write_tfrecord:image" + }, + "outputs": [], + "source": [ + "# Select TFRecord method of encoding\n", + "CACHE = TFExampleImageUncompressed # [ TFExampleImageUncompressed, TFExampleImagePreprocessed]\n", + "\n", + "# Map labels to class names\n", + "cls2label = {\"daisy\": 0, \"dandelion\": 1, \"roses\": 2, \"sunflowers\": 3, \"tulips\": 4}\n", + "\n", + "# Read in each example from exported dataset index\n", + "with tf.io.gfile.GFile(jsonl_index, \"r\") as f:\n", + " data = f.readlines()\n", + "\n", + "# The path to the TFRecord cached file.\n", + "GCS_TFRECORD_URI = BUCKET_NAME + \"/flowers.tfrecord\"\n", + "\n", + "# Create the TFRecord cached file\n", + "with tf.io.TFRecordWriter(GCS_TFRECORD_URI) as writer:\n", + " n = 0\n", + " for data_item in data:\n", + " j = json.loads(data_item)\n", + " image = j[\"imageGcsUri\"]\n", + " cls = j[\"classificationAnnotation\"][\"displayName\"]\n", + " label = cls2label[cls]\n", + " example = CACHE(image, label, shape=(128, 128))\n", + " writer.write(example.SerializeToString())\n", + " n += 1\n", + " if n % 10 == 0:\n", + " print(n, image)\n", + "\n", + "listing = ! gsutil ls -la $GCS_TFRECORD_URI\n", + "print(\"TFRecord File\", listing)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_model" + }, + "source": [ + "## Train a model\n", + "\n", + "There are two ways you can train a custom model using a container image:\n", + "\n", + "- **Use a Google Cloud prebuilt container**. If you use a prebuilt container, you will additionally specify a Python package to install into the container image. This Python package contains your code for training a custom model.\n", + "\n", + "- **Use your own custom container image**. If you use your own container, the container needs to contain your code for training a custom model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_specification:prebuilt_container" + }, + "source": [ + "## Prepare your custom job specification\n", + "\n", + "Now that your clients are ready, your first step is to create a Job Specification for your custom training job. The job specification will consist of the following:\n", + "\n", + "- `worker_pool_spec` : The specification of the type of machine(s) you will use for training and how many (single or distributed)\n", + "- `python_package_spec` : The specification of the Python package to be installed with the pre-built container." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "source": [ + "### Prepare your machine specification\n", + "\n", + "Now define the machine specification for your custom training job. This tells Vertex what type of machine instance to provision for the training.\n", + " - `machine_type`: The type of GCP instance to provision -- e.g., n1-standard-8.\n", + " - `accelerator_type`: The type, if any, of hardware accelerator. In this tutorial if you previously set the variable `TRAIN_GPU != None`, you are using a GPU; otherwise you will use a CPU.\n", + " - `accelerator_count`: The number of accelerators." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "outputs": [], + "source": [ + "if TRAIN_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": TRAIN_COMPUTE,\n", + " \"accelerator_type\": TRAIN_GPU,\n", + " \"accelerator_count\": TRAIN_NGPU,\n", + " }\n", + "else:\n", + " machine_spec = {\"machine_type\": TRAIN_COMPUTE, \"accelerator_count\": 0}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "source": [ + "### Prepare your disk specification\n", + "\n", + "(optional) Now define the disk specification for your custom training job. This tells Vertex what type and size of disk to provision in each machine instance for the training.\n", + "\n", + " - `boot_disk_type`: Either SSD or Standard. SSD is faster, and Standard is less expensive. Defaults to SSD.\n", + " - `boot_disk_size_gb`: Size of disk in GB." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "outputs": [], + "source": [ + "DISK_TYPE = \"pd-ssd\" # [ pd-ssd, pd-standard]\n", + "DISK_SIZE = 200 # GB\n", + "\n", + "disk_spec = {\"boot_disk_type\": DISK_TYPE, \"boot_disk_size_gb\": DISK_SIZE}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container,tfrecord" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_job_\" + TIMESTAMP\n", + "MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, JOB_NAME)\n", + "\n", + "if not TRAIN_NGPU or TRAIN_NGPU < 2:\n", + " TRAIN_STRATEGY = \"single\"\n", + "else:\n", + " TRAIN_STRATEGY = \"mirror\"\n", + "\n", + "EPOCHS = 20\n", + "STEPS = 100\n", + "\n", + "PARAM_FILE = BUCKET_NAME + \"/params.txt\"\n", + "\n", + "DIRECT = True\n", + "if DIRECT:\n", + " CMDARGS = [\n", + " \"--model-dir=\" + MODEL_DIR,\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "else:\n", + " CMDARGS = [\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "\n", + "worker_pool_spec = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": machine_spec,\n", + " \"disk_spec\": disk_spec,\n", + " \"python_package_spec\": {\n", + " \"executor_image_uri\": TRAIN_IMAGE,\n", + " \"package_uris\": [BUCKET_NAME + \"/trainer_flowers.tar.gz\"],\n", + " \"python_module\": \"trainer.task\",\n", + " \"args\": CMDARGS,\n", + " },\n", + " }\n", + "]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "assemble_custom_job_specification" + }, + "source": [ + "### Assemble a job specification\n", + "\n", + "Now assemble the complete description for the custom job specification:\n", + "\n", + "- `display_name`: The human readable name you assign to this custom job.\n", + "- `job_spec`: The specification for the custom job.\n", + " - `worker_pool_specs`: The specification for the machine VM instances.\n", + " - `base_output_directory`: This tells the service the Cloud Storage location where to save the model artifacts (when variable `DIRECT = False`). The service will then pass the location to the training script as the environment variable `AIP_MODEL_DIR`, and the path will be of the form:\n", + "\n", + " /model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "assemble_custom_job_specification" + }, + "outputs": [], + "source": [ + "if DIRECT:\n", + " job_spec = {\"worker_pool_specs\": worker_pool_spec}\n", + "else:\n", + " job_spec = {\n", + " \"worker_pool_specs\": worker_pool_spec,\n", + " \"base_output_directory\": {\"output_uri_prefix\": MODEL_DIR},\n", + " }\n", + "\n", + "custom_job = {\"display_name\": JOB_NAME, \"job_spec\": job_spec}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "examine_training_package" + }, + "source": [ + "### Examine the training package\n", + "\n", + "#### Package layout\n", + "\n", + "Before you start the training, you will look at how a Python package is assembled for a custom training job. When unarchived, the package contains the following directory/file layout.\n", + "\n", + "- PKG-INFO\n", + "- README.md\n", + "- setup.cfg\n", + "- setup.py\n", + "- trainer\n", + " - \\_\\_init\\_\\_.py\n", + " - task.py\n", + "\n", + "The files `setup.cfg` and `setup.py` are the instructions for installing the package into the operating environment of the Docker image.\n", + "\n", + "The file `trainer/task.py` is the Python script for executing the custom training job. *Note*, when we referred to it in the worker pool specification, we replace the directory slash with a dot (`trainer.task`) and dropped the file suffix (`.py`).\n", + "\n", + "#### Package Assembly\n", + "\n", + "In the following cells, you will assemble the training package." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "examine_training_package" + }, + "outputs": [], + "source": [ + "# Make folder for Python training script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\ntag_build =\\n\\ntag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'tensorflow_datasets==1.3.0',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\nName: Flowers image classification\\n\\nVersion: 0.0.0\\n\\nSummary: Demostration training script\\n\\nHome-page: www.google.com\\n\\nAuthor: Google\\n\\nAuthor-email: aferlitsch@google.com\\n\\nLicense: Public\\n\\nDescription: Demo\\n\\nPlatform: Vertex\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:flowers" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the training script task.py. We won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n", + "- Loads Flowers dataset from TFRecords (`args.training-data`)\n", + "- Creates a tf.data.Dataset generator from the TFRecord.\n", + "- Builds a simple ConvNet model using TF.Keras model API.\n", + "- Compiles the model (`compile()`).\n", + "- Sets a training distribution strategy for multi-workers using `args.distribute`.\n", + "- Trains the model (`fit()`) with epochs and steps according to the arguments `args.epochs` and `args.steps`\n", + "- Saves the trained model (`save(args.model_dir)`) to the specified model directory." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:flowers" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Single, Mirror and Multi-Machine Distributed Training for Flowers\n", + "\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "import argparse\n", + "import os\n", + "import sys\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default='/tmp/saved_model', type=str, help='Model dir.')\n", + "parser.add_argument('--training-data', dest='tfrecord_uri')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.001, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=10, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=100, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "print('DEVICES', device_lib.list_local_devices())\n", + "\n", + "# Single Machine, single compute device\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "# Single Machine, multiple compute device\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "# Multiple Machine, multiple compute device\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "# Multi-worker configuration\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "# Preparing dataset\n", + "BUFFER_SIZE = 10000\n", + "BATCH_SIZE = 64\n", + "\n", + "\n", + "def make_datasets_unbatched():\n", + " dataset = tf.data.TFRecordDataset(args.tfrecord_uri)\n", + "\n", + " feature_description = {\n", + " 'image': tf.io.FixedLenFeature([], tf.string),\n", + " 'label': tf.io.FixedLenFeature([], tf.int64),\n", + " 'shape': tf.io.FixedLenFeature([3], tf.int64),\n", + " }\n", + "\n", + "\n", + " def _parse_function(proto):\n", + " ''' parse the next serialized tf.train.Example using the feature description '''\n", + " example = tf.io.parse_single_example(proto, feature_description)\n", + " image = tf.io.decode_raw(example['image'], tf.int32)\n", + " shape = tf.cast(example['shape'], tf.int32)\n", + " label = tf.cast(example['label'], tf.int32)\n", + " image.set_shape([128 * 128 * 3])\n", + " image = tf.reshape(image, (128, 128, 3))\n", + " image = ( tf.cast(image, tf.float32) / 255.0)\n", + "\n", + " return image, label\n", + "\n", + "\n", + " return dataset.map(_parse_function).cache().shuffle(BUFFER_SIZE).repeat()\n", + "\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_cnn_model():\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu', input_shape=(128, 128, 3)),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Conv2D(64, 3, activation='relu'),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Conv2D(128, 3, activation='relu'),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Flatten(),\n", + " tf.keras.layers.Dense(5, activation='softmax')\n", + " ])\n", + " model.compile(\n", + " loss=tf.keras.losses.sparse_categorical_crossentropy,\n", + " optimizer=tf.keras.optimizers.SGD(learning_rate=args.lr),\n", + " metrics=['accuracy'])\n", + " return model\n", + "\n", + "\n", + "with strategy.scope():\n", + " # Creation of dataset, and model building/compiling need to be within\n", + " # `strategy.scope()`.\n", + " model = build_and_compile_cnn_model()\n", + "\n", + "# Train the model\n", + "NUM_WORKERS = strategy.num_replicas_in_sync\n", + "# Here the batch size scales up by number of workers since\n", + "# `tf.data.Dataset.batch` expects the global batch size.\n", + "GLOBAL_BATCH_SIZE = BATCH_SIZE * NUM_WORKERS\n", + "train_dataset = make_datasets_unbatched().batch(GLOBAL_BATCH_SIZE)\n", + "\n", + "model.fit(x=train_dataset, epochs=args.epochs, steps_per_epoch=args.steps)\n", + "model.save(args.model_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tarball_training_script" + }, + "source": [ + "#### Store training script on your Cloud Storage bucket\n", + "\n", + "Next, you package the training folder into a compressed tar ball, and then store it in your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tarball_training_script" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_flowers.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job" + }, + "source": [ + "### Train the model\n", + "\n", + "\n", + "Now start the training of your custom training job on Vertex. Use this helper function `create_custom_job`, which takes the following parameter:\n", + "\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "The helper function calls job client service's `create_custom_job` method, with the following parameters:\n", + "\n", + "-`parent`: The Vertex location path to `Dataset`, `Model` and `Endpoint` resources.\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "You will display a handful of the fields returned in `response` object, with the two that are of most interest are:\n", + "\n", + "`response.name`: The Vertex fully qualified identifier assigned to this custom training job. You save this identifier for using in subsequent steps.\n", + "\n", + "`response.state`: The current state of the custom training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job" + }, + "outputs": [], + "source": [ + "def create_custom_job(custom_job):\n", + " response = clients[\"job\"].create_custom_job(parent=PARENT, custom_job=custom_job)\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = create_custom_job(custom_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "job_id:response" + }, + "source": [ + "Now get the unique identifier for the custom job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom job\n", + "job_id = response.name\n", + "# The short numeric ID for the custom job\n", + "job_short_id = job_id.split(\"/\")[-1]\n", + "\n", + "print(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_custom_job" + }, + "source": [ + "### Get information on a custom job\n", + "\n", + "Next, use this helper function `get_custom_job`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "The helper function calls the job client service's`get_custom_job` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "If you recall, you got the Vertex fully qualified identifier for the custom job in the `response.name` field when you called the `create_custom_job` method, and saved the identifier in the variable `job_id`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_custom_job" + }, + "outputs": [], + "source": [ + "def get_custom_job(name, silent=False):\n", + " response = clients[\"job\"].get_custom_job(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = get_custom_job(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete:custom" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, we will need to know the location of the saved model, which the Python script saved in your local Cloud Storage bucket at `MODEL_DIR + '/saved_model.pb'`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete:custom" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_custom_job(job_id, True)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_path_to_deploy = None\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " if not DIRECT:\n", + " MODEL_DIR = MODEL_DIR + \"/model\"\n", + " model_path_to_deploy = MODEL_DIR\n", + " print(\"Training Time:\", response.update_time - response.create_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model_to_deploy:\", model_path_to_deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "load_saved_model" + }, + "source": [ + "## Load the saved model\n", + "\n", + "Your model is stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Now load it from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n", + "\n", + "To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "load_saved_model" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:image" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load the Flowers test (holdout) data from `tf.keras.datasets`, using the method `load_data()`. This will return the dataset as a tuple of two elements. The first element is the training data and the second is the test data. Each element is also a tuple of two elements: the image data, and the corresponding labels.\n", + "\n", + "You don't need the training data, and hence why we loaded it as `(_, _)`.\n", + "\n", + "Before you can run the data through evaluation, you need to preprocess it:\n", + "\n", + "x_test:\n", + "1. Normalize (rescaling) the pixel data by dividing each pixel by 255. This will replace each single byte integer pixel with a 32-bit floating point number between 0 and 1.\n", + "\n", + "y_test:
\n", + "2. The labels are currently scalar (sparse). If you look back at the `compile()` step in the `trainer/task.py` script, you will find that it was compiled for sparse labels. So we don't need to do anything more." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:image,exported_ds" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now let's find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load some sample data from the end of the Flowers dataset, last 10 items. We will then preprocessed the data items to form:\n", + "\n", + "- `x_test`: The preprocessed image data in memory.\n", + "- `y_test`: The corresponding labels." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "evaluate_custom_model:image,exported_ds" + }, + "outputs": [], + "source": [ + "x_test = []\n", + "y_test = []\n", + "\n", + "data_items = export_data_items[-10:]\n", + "for data_item in data_items:\n", + " data_item = json.loads(data_item)\n", + " print(\"FILE\", data_item[\"imageGcsUri\"])\n", + " with tf.io.gfile.GFile(data_item[\"imageGcsUri\"], \"rb\") as f:\n", + " data = f.read()\n", + " image = tf.io.decode_image(data)\n", + " image = tf.image.resize(image, (128, 128))\n", + " image = (image.numpy() / 255.0).astype(np.float32)\n", + " cls = data_item[\"classificationAnnotation\"][\"displayName\"]\n", + " label = cls2label[cls]\n", + " x_test.append(image)\n", + " y_test.append(label)\n", + "\n", + "x_test = np.asarray(x_test)\n", + "y_test = np.asarray(y_test)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "perform_evaluation_custom" + }, + "source": [ + "### Perform the model evaluation\n", + "\n", + "Now evaluate how well the model in the custom job did." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "perform_evaluation_custom" + }, + "outputs": [], + "source": [ + "model.evaluate(x_test, y_test)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "how_serving_function_works" + }, + "source": [ + "## Upload the model for serving\n", + "\n", + "Next, you will upload your TF.Keras model from the custom job to Vertex `Model` service, which will create a Vertex `Model` resource for your custom model. During upload, you need to define a serving function to convert data to the format your model expects. If you send encoded data to Vertex, your serving function ensures that the data is decoded on the model server before it is passed as input to your model.\n", + "\n", + "### How does the serving function work\n", + "\n", + "When you send a request to an online prediction server, the request is received by a HTTP server. The HTTP server extracts the prediction request from the HTTP request content body. The extracted prediction request is forwarded to the serving function. For Google pre-built prediction containers, the request content is passed to the serving function as a `tf.string`.\n", + "\n", + "The serving function consists of two parts:\n", + "\n", + "- `preprocessing function`:\n", + " - Converts the input (`tf.string`) to the input shape and data type of the underlying model (dynamic graph).\n", + " - Performs the same preprocessing of the data that was done during training the underlying model -- e.g., normalizing, scaling, etc.\n", + "- `post-processing function`:\n", + " - Converts the model output to format expected by the receiving application -- e.q., compresses the output.\n", + " - Packages the output for the the receiving application -- e.g., add headings, make JSON object, etc.\n", + "\n", + "Both the preprocessing and post-processing functions are converted to static graphs which are fused to the model. The output from the underlying model is passed to the post-processing function. The post-processing function passes the converted/packaged output back to the HTTP server. The HTTP server returns the output as the HTTP response content.\n", + "\n", + "One consideration you need to consider when building serving functions for TF.Keras models is that they run as static graphs. That means, you cannot use TF graph operations that require a dynamic graph. If you do, you will get an error during the compile of the serving function which will indicate that you are using an EagerTensor which is not supported." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_image" + }, + "source": [ + "### Serving function for image data\n", + "\n", + "To pass images to the prediction service, you encode the compressed (e.g., JPEG) image bytes into base 64 -- which makes the content safe from modification while transmitting binary data over the network. Since this deployed model expects input data as raw (uncompressed) bytes, you need to ensure that the base 64 encoded data gets converted back to raw bytes before it is passed as input to the deployed model.\n", + "\n", + "To resolve this, define a serving function (`serving_fn`) and attach it to the model as a preprocessing step. Add a `@tf.function` decorator so the serving function is fused to the underlying model (instead of upstream on a CPU).\n", + "\n", + "When you send a prediction or explanation request, the content of the request is base 64 decoded into a Tensorflow string (`tf.string`), which is passed to the serving function (`serving_fn`). The serving function preprocesses the `tf.string` into raw (uncompressed) numpy bytes (`preprocess_fn`) to match the input requirements of the model:\n", + "- `io.decode_jpeg`- Decompresses the JPG image which is returned as a Tensorflow tensor with three channels (RGB).\n", + "- `image.convert_image_dtype` - Changes integer pixel values to float 32.\n", + "- `image.resize` - Resizes the image to match the input shape for the model.\n", + "- `resized / 255.0` - Rescales (normalization) the pixel data between 0 and 1.\n", + "\n", + "At this point, the data can be passed to the model (`m_call`)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_image" + }, + "outputs": [], + "source": [ + "CONCRETE_INPUT = \"numpy_inputs\"\n", + "\n", + "\n", + "def _preprocess(bytes_input):\n", + " decoded = tf.io.decode_jpeg(bytes_input, channels=3)\n", + " decoded = tf.image.convert_image_dtype(decoded, tf.float32)\n", + " resized = tf.image.resize(decoded, size=(128, 128))\n", + " rescale = tf.cast(resized / 255.0, tf.float32)\n", + " return rescale\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def preprocess_fn(bytes_inputs):\n", + " decoded_images = tf.map_fn(\n", + " _preprocess, bytes_inputs, dtype=tf.float32, back_prop=False\n", + " )\n", + " return {\n", + " CONCRETE_INPUT: decoded_images\n", + " } # User needs to make sure the key matches model's input\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def serving_fn(bytes_inputs):\n", + " images = preprocess_fn(bytes_inputs)\n", + " prob = m_call(**images)\n", + " return prob\n", + "\n", + "\n", + "m_call = tf.function(model.call).get_concrete_function(\n", + " [tf.TensorSpec(shape=[None, 128, 128, 3], dtype=tf.float32, name=CONCRETE_INPUT)]\n", + ")\n", + "\n", + "tf.saved_model.save(\n", + " model, model_path_to_deploy, signatures={\"serving_default\": serving_fn}\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_signature:image" + }, + "source": [ + "## Get the serving function signature\n", + "\n", + "You can get the signatures of your model's input and output layers by reloading the model into memory, and querying it for the signatures corresponding to each layer.\n", + "\n", + "For your purpose, you need the signature of the serving function. Why? Well, when we send our data for prediction as a HTTP request packet, the image data is base64 encoded, and our TF.Keras model takes numpy input. Your serving function will do the conversion from base64 to a numpy array.\n", + "\n", + "When making a prediction request, you need to route the request to the serving function instead of the model, so you need to know the input layer name of the serving function -- which you will use later when you make a prediction request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_signature:image" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_path_to_deploy)\n", + "\n", + "serving_input = list(\n", + " loaded.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "print(\"Serving function input:\", serving_input)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "upload_the_model" + }, + "source": [ + "### Upload the model\n", + "\n", + "Use this helper function `upload_model` to upload your model, stored in SavedModel format, up to the `Model` service, which will instantiate a Vertex `Model` resource instance for your model. Once you've done that, you can use the `Model` resource instance in the same way as any other Vertex `Model` resource instance, such as deploying to an `Endpoint` resource for serving predictions.\n", + "\n", + "The helper function takes the following parameters:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` service.\n", + "- `image_uri`: The container image for the model deployment.\n", + "- `model_uri`: The Cloud Storage path to our SavedModel artificat. For this tutorial, this is the Cloud Storage location where the `trainer/task.py` saved the model artifacts, which we specified in the variable `MODEL_DIR`.\n", + "\n", + "The helper function calls the `Model` client service's method `upload_model`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for `Dataset`, `Model` and `Endpoint` resources.\n", + "- `model`: The specification for the Vertex `Model` resource instance.\n", + "\n", + "Let's now dive deeper into the Vertex model specification `model`. This is a dictionary object that consists of the following fields:\n", + "\n", + "- `display_name`: A human readable name for the `Model` resource.\n", + "- `metadata_schema_uri`: Since your model was built without an Vertex `Dataset` resource, you will leave this blank (`''`).\n", + "- `artificat_uri`: The Cloud Storage path where the model is stored in SavedModel format.\n", + "- `container_spec`: This is the specification for the Docker container that will be installed on the `Endpoint` resource, from which the `Model` resource will serve predictions. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + "\n", + "Uploading a model into a Vertex Model resource returns a long running operation, since it may take a few moments. You call response.result(), which is a synchronous call and will return when the Vertex Model resource is ready.\n", + "\n", + "The helper function returns the Vertex fully qualified identifier for the corresponding Vertex Model instance upload_model_response.model. You will save the identifier for subsequent steps in the variable model_to_deploy_id." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "upload_the_model" + }, + "outputs": [], + "source": [ + "IMAGE_URI = DEPLOY_IMAGE\n", + "\n", + "\n", + "def upload_model(display_name, image_uri, model_uri):\n", + " model = {\n", + " \"display_name\": display_name,\n", + " \"metadata_schema_uri\": \"\",\n", + " \"artifact_uri\": model_uri,\n", + " \"container_spec\": {\n", + " \"image_uri\": image_uri,\n", + " \"command\": [],\n", + " \"args\": [],\n", + " \"env\": [{\"name\": \"env_name\", \"value\": \"env_value\"}],\n", + " \"ports\": [{\"container_port\": 8080}],\n", + " \"predict_route\": \"\",\n", + " \"health_route\": \"\",\n", + " },\n", + " }\n", + " response = clients[\"model\"].upload_model(parent=PARENT, model=model)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " upload_model_response = response.result(timeout=180)\n", + " print(\"upload_model_response\")\n", + " print(\" model:\", upload_model_response.model)\n", + " return upload_model_response.model\n", + "\n", + "\n", + "model_to_deploy_id = upload_model(\n", + " \"flowers-\" + TIMESTAMP, IMAGE_URI, model_path_to_deploy\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_model" + }, + "source": [ + "### Get `Model` resource information\n", + "\n", + "Now let's get the model information for just your model. Use this helper function `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource.\n", + "\n", + "This helper function calls the Vertex `Model` client service's method `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_model" + }, + "outputs": [], + "source": [ + "def get_model(name):\n", + " response = clients[\"model\"].get_model(name=name)\n", + " print(response)\n", + "\n", + "\n", + "get_model(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:custom" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex custom `Model` resource. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"flowers_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:dedicated" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `dedicated_resources`: This refers to how many compute instances (replicas) that are scaled for serving prediction requests.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `min_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:dedicated" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"flowers_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " \"machine_spec\": machine_spec,\n", + " },\n", + " \"disable_container_logging\": False,\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:exported_ds" + }, + "outputs": [], + "source": [ + "# Last data item in exported dataset index\n", + "data_items = export_data_items[-1:]\n", + "data_item = json.loads(data_items[0])\n", + "image_path = data_item[\"imageGcsUri\"]\n", + "print(\"IMAGE PATH\", image_path)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "prepare_test_item:exported_ds" + }, + "source": [ + "### Prepare the request content\n", + "You are going to send the image as compressed JPG image, instead of the raw uncompressed bytes:\n", + "\n", + "- `cv2.imread`: Read the compressed JPG images back into memory as raw bytes.\n", + "- `tf.image.resize`: Resize the image to the input shape of the model -- (128, 128, 3).\n", + "- cv2.imwrite: Write resized image back to disk.\n", + "- `base64.b64encode`: Read back the resized compressed image and encode into a base 64 encoded string." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "prepare_test_item:exported_ds" + }, + "outputs": [], + "source": [ + "! gsutil cp $image_path tmp.jpg\n", + "\n", + "import base64\n", + "\n", + "import cv2\n", + "\n", + "test_image = cv2.imread(\"tmp.jpg\", cv2.IMREAD_COLOR)\n", + "print(\"before:\", test_image.shape)\n", + "test_image = cv2.resize(test_image, (128, 128))\n", + "print(\"after:\", test_image.shape)\n", + "cv2.imwrite(\"tmp.jpg\", test_image.astype(np.uint8))\n", + "\n", + "# bytes = tf.io.read_file('tmp.jpg')\n", + "with open(\"tmp.jpg\", \"rb\") as f:\n", + " bytes = f.read()\n", + "b64str = base64.b64encode(bytes).decode(\"utf-8\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "send_prediction_request:image" + }, + "source": [ + "### Send the prediction request\n", + "\n", + "Ok, now you have a test image. Use this helper function `predict_image`, which takes the following parameters:\n", + "\n", + "- `image`: The test image data as a numpy array.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed to.\n", + "- `parameters_dict`: Additional parameters for serving.\n", + "\n", + "This function calls the prediction client service `predict` method with the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed to.\n", + "- `instances`: A list of instances (encoded images) to predict.\n", + "- `parameters`: Additional parameters for serving.\n", + "\n", + "To pass the image data to the prediction service, in the previous step you encoded the bytes into base64 -- which makes the content safe from modification when transmitting binary data over the network. You need to tell the serving binary where your model is deployed to, that the content has been base64 encoded, so it will decode it on the other end in the serving binary.\n", + "\n", + "Each instance in the prediction request is a dictionary entry of the form:\n", + "\n", + " {serving_input: {'b64': content}}\n", + "\n", + "- `input_name`: the name of the input layer of the underlying model.\n", + "- `'b64'`: A key that indicates the content is base64 encoded.\n", + "- `content`: The compressed JPG image bytes as a base64 encoded string.\n", + "\n", + "Since the `predict()` service can take multiple images (instances), you will send your single image as a list of one image. As a final step, you package the instances list into Google's protobuf format -- which is what we pass to the `predict()` service.\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding image in the request. You will see in the output for each prediction:\n", + "\n", + "- `predictions`: Confidence level for the prediction, between 0 and 1, for each of the classes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "send_prediction_request:image" + }, + "outputs": [], + "source": [ + "def predict_image(image, endpoint, parameters_dict):\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [{serving_input: {\"b64\": image}}]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters_dict\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", prediction)\n", + "\n", + "\n", + "predict_image(b64str, endpoint_id, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_custom_image_classification_online_exported_ds.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_custom_image_classification_online_pipeline.ipynb b/notebooks/community/gapic/custom/showcase_custom_image_classification_online_pipeline.ipynb new file mode 100644 index 000000000..4b4956f19 --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_custom_image_classification_online_pipeline.ipynb @@ -0,0 +1,2261 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Custom training image classification model with pipeline for online prediction with training pipeline\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom,training_pipeline" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to train and deploy a custom image classification model for online prediction, using a training pipeline." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:custom,cifar10,icn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [CIFAR10 dataset](https://www.tensorflow.org/datasets/catalog/cifar10) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use is built into TensorFlow. The trained model predicts which type of class an image is from ten classes: airplane, automobile, bird, cat, deer, dog, frog, horse, ship, truck." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,training,online_prediction,training_pipeline" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create a custom model from a Python script in a Google prebuilt Docker container using the Vertex client library, and then do a prediction on the deployed model by sending data. You can alternatively create custom models using `gcloud` command-line tool or online using Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex custom job for training a model.\n", + "- Create a `TrainingPipeline` resource.\n", + "- Train a TensorFlow model with the `TrainingPipeline` resource.\n", + "- Retrieve and load the model artifacts.\n", + "- View the model evaluation.\n", + "- Upload the model as a Vertex `Model` resource.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "custom_constants" + }, + "source": [ + "#### CustomJob constants\n", + "\n", + "Set constants unique to CustomJob training:\n", + "\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "custom_constants" + }, + "outputs": [], + "source": [ + "CUSTOM_TASK_GCS_PATH = (\n", + " \"gs://google-cloud-aiplatform/schema/trainingjob/definition/custom_task_1.0.0.yaml\"\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for training and prediction.\n", + "\n", + "Set the variables `TRAIN_GPU/TRAIN_NGPU` and `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU.\n", + "\n", + "*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3 -- which is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n", + " TRAIN_GPU, TRAIN_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n", + " )\n", + "else:\n", + " TRAIN_GPU, TRAIN_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (None, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:training,prediction" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for training and prediction\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-3:latest`\n", + " - TensorFlow 2.4\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-4:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/training/xgboost-cpu.1-1`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/training/scikit-learn-cpu.0-23:latest`\n", + " - Pytorch\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-5:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-6:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-7:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for training](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers).\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-3:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-90:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-82:latest`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-23:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-22:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-20:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for prediction](https://cloud.google.com/vertex-ai/docs/predictions/pre-built-containers)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf2-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf2-cpu.{}\".format(TF)\n", + "else:\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "TRAIN_IMAGE = \"gcr.io/cloud-aiplatform/training/{}:latest\".format(TRAIN_VERSION)\n", + "DEPLOY_IMAGE = \"gcr.io/cloud-aiplatform/prediction/{}:latest\".format(DEPLOY_VERSION)\n", + "\n", + "print(\"Training:\", TRAIN_IMAGE, TRAIN_GPU, TRAIN_NGPU)\n", + "print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU, DEPLOY_NGPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:training,prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for training and prediction.\n", + "\n", + "- Set the variables `TRAIN_COMPUTE` and `DEPLOY_COMPUTE` to configure the compute resources for the VMs you will use for for training and prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: The following is not supported for training:*\n", + "\n", + " - `standard`: 2 vCPUs\n", + " - `highcpu`: 2, 4 and 8 vCPUs\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_TRAIN_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Train machine type\", TRAIN_COMPUTE)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own custom model and training for CIFAR10." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:custom,training_pipeline" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:custom,training_pipeline" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_model" + }, + "source": [ + "## Train a model\n", + "\n", + "There are two ways you can train a custom model using a container image:\n", + "\n", + "- **Use a Google Cloud prebuilt container**. If you use a prebuilt container, you will additionally specify a Python package to install into the container image. This Python package contains your code for training a custom model.\n", + "\n", + "- **Use your own custom container image**. If you use your own container, the container needs to contain your code for training a custom model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_specification:prebuilt_container" + }, + "source": [ + "## Prepare your custom job specification\n", + "\n", + "Now that your clients are ready, your first step is to create a Job Specification for your custom training job. The job specification will consist of the following:\n", + "\n", + "- `worker_pool_spec` : The specification of the type of machine(s) you will use for training and how many (single or distributed)\n", + "- `python_package_spec` : The specification of the Python package to be installed with the pre-built container." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "source": [ + "### Prepare your machine specification\n", + "\n", + "Now define the machine specification for your custom training job. This tells Vertex what type of machine instance to provision for the training.\n", + " - `machine_type`: The type of GCP instance to provision -- e.g., n1-standard-8.\n", + " - `accelerator_type`: The type, if any, of hardware accelerator. In this tutorial if you previously set the variable `TRAIN_GPU != None`, you are using a GPU; otherwise you will use a CPU.\n", + " - `accelerator_count`: The number of accelerators." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "outputs": [], + "source": [ + "if TRAIN_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": TRAIN_COMPUTE,\n", + " \"accelerator_type\": TRAIN_GPU,\n", + " \"accelerator_count\": TRAIN_NGPU,\n", + " }\n", + "else:\n", + " machine_spec = {\"machine_type\": TRAIN_COMPUTE, \"accelerator_count\": 0}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "source": [ + "### Prepare your disk specification\n", + "\n", + "(optional) Now define the disk specification for your custom training job. This tells Vertex what type and size of disk to provision in each machine instance for the training.\n", + "\n", + " - `boot_disk_type`: Either SSD or Standard. SSD is faster, and Standard is less expensive. Defaults to SSD.\n", + " - `boot_disk_size_gb`: Size of disk in GB." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "outputs": [], + "source": [ + "DISK_TYPE = \"pd-ssd\" # [ pd-ssd, pd-standard]\n", + "DISK_SIZE = 200 # GB\n", + "\n", + "disk_spec = {\"boot_disk_type\": DISK_TYPE, \"boot_disk_size_gb\": DISK_SIZE}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container" + }, + "source": [ + "### Define the worker pool specification\n", + "\n", + "Next, you define the worker pool specification for your custom training job. The worker pool specification will consist of the following:\n", + "\n", + "- `replica_count`: The number of instances to provision of this machine type.\n", + "- `machine_spec`: The hardware specification.\n", + "- `disk_spec` : (optional) The disk storage specification.\n", + "\n", + "- `python_package`: The Python training package to install on the VM instance(s) and which Python module to invoke, along with command line arguments for the Python module.\n", + "\n", + "Let's dive deeper now into the python package specification:\n", + "\n", + "-`executor_image_spec`: This is the docker image which is configured for your custom training job.\n", + "\n", + "-`package_uris`: This is a list of the locations (URIs) of your python training packages to install on the provisioned instance. The locations need to be in a Cloud Storage bucket. These can be either individual python files or a zip (archive) of an entire package. In the later case, the job service will unzip (unarchive) the contents into the docker image.\n", + "\n", + "-`python_module`: The Python module (script) to invoke for running the custom training job. In this example, you will be invoking `trainer.task.py` -- note that it was not neccessary to append the `.py` suffix.\n", + "\n", + "-`args`: The command line arguments to pass to the corresponding Pythom module. In this example, you will be setting:\n", + " - `\"--model-dir=\" + MODEL_DIR` : The Cloud Storage location where to store the model artifacts. There are two ways to tell the training script where to save the model artifacts:\n", + " - direct: You pass the Cloud Storage location as a command line argument to your training script (set variable `DIRECT = True`), or\n", + " - indirect: The service passes the Cloud Storage location as the environment variable `AIP_MODEL_DIR` to your training script (set variable `DIRECT = False`). In this case, you tell the service the model artifact location in the job specification.\n", + " - `\"--epochs=\" + EPOCHS`: The number of epochs for training.\n", + " - `\"--steps=\" + STEPS`: The number of steps (batches) per epoch.\n", + " - `\"--distribute=\" + TRAIN_STRATEGY\"` : The training distribution strategy to use for single or distributed training.\n", + " - `\"single\"`: single device.\n", + " - `\"mirror\"`: all GPU devices on a single compute instance.\n", + " - `\"multi\"`: all GPU devices on all compute instances." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_job_\" + TIMESTAMP\n", + "MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, JOB_NAME)\n", + "\n", + "if not TRAIN_NGPU or TRAIN_NGPU < 2:\n", + " TRAIN_STRATEGY = \"single\"\n", + "else:\n", + " TRAIN_STRATEGY = \"mirror\"\n", + "\n", + "EPOCHS = 20\n", + "STEPS = 100\n", + "\n", + "DIRECT = True\n", + "if DIRECT:\n", + " CMDARGS = [\n", + " \"--model-dir=\" + MODEL_DIR,\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "else:\n", + " CMDARGS = [\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "\n", + "worker_pool_spec = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": machine_spec,\n", + " \"disk_spec\": disk_spec,\n", + " \"python_package_spec\": {\n", + " \"executor_image_uri\": TRAIN_IMAGE,\n", + " \"package_uris\": [BUCKET_NAME + \"/trainer_cifar10.tar.gz\"],\n", + " \"python_module\": \"trainer.task\",\n", + " \"args\": CMDARGS,\n", + " },\n", + " }\n", + "]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "examine_training_package" + }, + "source": [ + "### Examine the training package\n", + "\n", + "#### Package layout\n", + "\n", + "Before you start the training, you will look at how a Python package is assembled for a custom training job. When unarchived, the package contains the following directory/file layout.\n", + "\n", + "- PKG-INFO\n", + "- README.md\n", + "- setup.cfg\n", + "- setup.py\n", + "- trainer\n", + " - \\_\\_init\\_\\_.py\n", + " - task.py\n", + "\n", + "The files `setup.cfg` and `setup.py` are the instructions for installing the package into the operating environment of the Docker image.\n", + "\n", + "The file `trainer/task.py` is the Python script for executing the custom training job. *Note*, when we referred to it in the worker pool specification, we replace the directory slash with a dot (`trainer.task`) and dropped the file suffix (`.py`).\n", + "\n", + "#### Package Assembly\n", + "\n", + "In the following cells, you will assemble the training package." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "examine_training_package" + }, + "outputs": [], + "source": [ + "# Make folder for Python training script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\ntag_build =\\n\\ntag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'tensorflow_datasets==1.3.0',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\nName: CIFAR10 image classification\\n\\nVersion: 0.0.0\\n\\nSummary: Demostration training script\\n\\nHome-page: www.google.com\\n\\nAuthor: Google\\n\\nAuthor-email: aferlitsch@google.com\\n\\nLicense: Public\\n\\nDescription: Demo\\n\\nPlatform: Vertex\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:cifar10" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the training script task.py. We won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n", + "- Loads CIFAR10 dataset from TF Datasets (tfds).\n", + "- Builds a model using TF.Keras model API.\n", + "- Compiles the model (`compile()`).\n", + "- Sets a training distribution strategy according to the argument `args.distribute`.\n", + "- Trains the model (`fit()`) with epochs and steps according to the arguments `args.epochs` and `args.steps`\n", + "- Saves the trained model (`save(args.model_dir)`) to the specified model directory." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:cifar10" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Single, Mirror and Multi-Machine Distributed Training for CIFAR-10\n", + "\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "import argparse\n", + "import os\n", + "import sys\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv(\"AIP_MODEL_DIR\"), type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.01, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=10, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=200, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "print('DEVICES', device_lib.list_local_devices())\n", + "\n", + "# Single Machine, single compute device\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "# Single Machine, multiple compute device\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "# Multiple Machine, multiple compute device\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "# Multi-worker configuration\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "# Preparing dataset\n", + "BUFFER_SIZE = 10000\n", + "BATCH_SIZE = 64\n", + "\n", + "\n", + "def make_datasets_unbatched():\n", + "\n", + " # Scaling CIFAR10 data from (0, 255] to (0., 1.]\n", + " def scale(image, label):\n", + " image = tf.cast(image, tf.float32)\n", + " image /= 255.0\n", + " return image, label\n", + "\n", + "\n", + " datasets, info = tfds.load(name='cifar10',\n", + " with_info=True,\n", + " as_supervised=True)\n", + " return datasets['train'].map(scale).cache().shuffle(BUFFER_SIZE).repeat()\n", + "\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_cnn_model():\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu', input_shape=(32, 32, 3)),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu'),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Flatten(),\n", + " tf.keras.layers.Dense(10, activation='softmax')\n", + " ])\n", + " model.compile(\n", + " loss=tf.keras.losses.sparse_categorical_crossentropy,\n", + " optimizer=tf.keras.optimizers.SGD(learning_rate=args.lr),\n", + " metrics=['accuracy'])\n", + " return model\n", + "\n", + "\n", + "# Train the model\n", + "NUM_WORKERS = strategy.num_replicas_in_sync\n", + "# Here the batch size scales up by number of workers since\n", + "# `tf.data.Dataset.batch` expects the global batch size.\n", + "GLOBAL_BATCH_SIZE = BATCH_SIZE * NUM_WORKERS\n", + "train_dataset = make_datasets_unbatched().batch(GLOBAL_BATCH_SIZE)\n", + "\n", + "with strategy.scope():\n", + " # Creation of dataset, and model building/compiling need to be within\n", + " # `strategy.scope()`.\n", + " model = build_and_compile_cnn_model()\n", + "\n", + "model.fit(x=train_dataset, epochs=args.epochs, steps_per_epoch=args.steps)\n", + "model.save(args.model_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tarball_training_script" + }, + "source": [ + "#### Store training script on your Cloud Storage bucket\n", + "\n", + "Next, you package the training folder into a compressed tar ball, and then store it in your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tarball_training_script" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_cifar10.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_with_pipeline" + }, + "source": [ + "## Train the model using a `TrainingPipeline` resource\n", + "\n", + "Now start training of your custom training job using a training pipeline on Vertex. To train the your custom model, do the following steps:\n", + "\n", + "1. Create a Vertex `TrainingPipeline` resource for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training.\n", + "\n", + "### Create a `TrainingPipeline` resource\n", + "\n", + "You may ask, what do we use a pipeline for? We typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "#### The `training_pipeline` specification\n", + "\n", + "First, you need to describe a pipeline specification. Let's look into the *minimal* requirements for constructing a `training_pipeline` specification for a custom job:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The training task schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A dictionary describing the specification for the (uploaded) Vertex custom `Model` resource.\n", + " - `display_name`: A human readable name for the `Model` resource.\n", + " - `artificat_uri`: The Cloud Storage path where the model artifacts are stored in SavedModel format.\n", + " - `container_spec`: This is the specification for the Docker container that will be installed on the `Endpoint` resource, from which the custom model will serve predictions." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_with_pipeline" + }, + "outputs": [], + "source": [ + "from google.protobuf import json_format\n", + "from google.protobuf.struct_pb2 import Value\n", + "\n", + "MODEL_NAME = \"custom_pipeline-\" + TIMESTAMP\n", + "PIPELINE_DISPLAY_NAME = \"custom-training-pipeline\" + TIMESTAMP\n", + "\n", + "training_task_inputs = json_format.ParseDict(\n", + " {\"workerPoolSpecs\": worker_pool_spec}, Value()\n", + ")\n", + "pipeline = {\n", + " \"display_name\": PIPELINE_DISPLAY_NAME,\n", + " \"training_task_definition\": CUSTOM_TASK_GCS_PATH,\n", + " \"training_task_inputs\": training_task_inputs,\n", + " \"model_to_upload\": {\n", + " \"display_name\": PIPELINE_DISPLAY_NAME + \"-model\",\n", + " \"artifact_uri\": MODEL_DIR,\n", + " \"container_spec\": {\"image_uri\": DEPLOY_IMAGE},\n", + " },\n", + "}\n", + "\n", + "print(pipeline)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_training_pipeline:custom" + }, + "source": [ + "#### Create the training pipeline\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameter:\n", + "\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "The helper function calls the pipeline client service's `create_pipeline` method, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: The full specification for the pipeline training job.\n", + "\n", + "The helper function will return the Vertex fully qualified identifier assigned to the training pipeline, which is saved as `pipeline.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_training_pipeline:custom" + }, + "outputs": [], + "source": [ + "def create_pipeline(training_pipeline):\n", + "\n", + " try:\n", + " pipeline = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " )\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline\n", + "\n", + "\n", + "response = create_pipeline(pipeline)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "model_path_to_deploy" + }, + "outputs": [], + "source": [ + "if not DIRECT:\n", + " MODEL_DIR = MODEL_DIR + \"/model\"\n", + "model_path_to_deploy = MODEL_DIR" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "load_saved_model" + }, + "source": [ + "## Load the saved model\n", + "\n", + "Your model is stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Now load it from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n", + "\n", + "To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "load_saved_model" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:image" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load the CIFAR10 test (holdout) data from `tf.keras.datasets`, using the method `load_data()`. This will return the dataset as a tuple of two elements. The first element is the training data and the second is the test data. Each element is also a tuple of two elements: the image data, and the corresponding labels.\n", + "\n", + "You don't need the training data, and hence why we loaded it as `(_, _)`.\n", + "\n", + "Before you can run the data through evaluation, you need to preprocess it:\n", + "\n", + "x_test:\n", + "1. Normalize (rescaling) the pixel data by dividing each pixel by 255. This will replace each single byte integer pixel with a 32-bit floating point number between 0 and 1.\n", + "\n", + "y_test:
\n", + "2. The labels are currently scalar (sparse). If you look back at the `compile()` step in the `trainer/task.py` script, you will find that it was compiled for sparse labels. So we don't need to do anything more." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "evaluate_custom_model:image,cifar10" + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "from tensorflow.keras.datasets import cifar10\n", + "\n", + "(_, _), (x_test, y_test) = cifar10.load_data()\n", + "x_test = (x_test / 255.0).astype(np.float32)\n", + "\n", + "print(x_test.shape, y_test.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "perform_evaluation_custom" + }, + "source": [ + "### Perform the model evaluation\n", + "\n", + "Now evaluate how well the model in the custom job did." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "perform_evaluation_custom" + }, + "outputs": [], + "source": [ + "model.evaluate(x_test, y_test)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "how_serving_function_works" + }, + "source": [ + "## Upload the model for serving\n", + "\n", + "Next, you will upload your TF.Keras model from the custom job to Vertex `Model` service, which will create a Vertex `Model` resource for your custom model. During upload, you need to define a serving function to convert data to the format your model expects. If you send encoded data to Vertex, your serving function ensures that the data is decoded on the model server before it is passed as input to your model.\n", + "\n", + "### How does the serving function work\n", + "\n", + "When you send a request to an online prediction server, the request is received by a HTTP server. The HTTP server extracts the prediction request from the HTTP request content body. The extracted prediction request is forwarded to the serving function. For Google pre-built prediction containers, the request content is passed to the serving function as a `tf.string`.\n", + "\n", + "The serving function consists of two parts:\n", + "\n", + "- `preprocessing function`:\n", + " - Converts the input (`tf.string`) to the input shape and data type of the underlying model (dynamic graph).\n", + " - Performs the same preprocessing of the data that was done during training the underlying model -- e.g., normalizing, scaling, etc.\n", + "- `post-processing function`:\n", + " - Converts the model output to format expected by the receiving application -- e.q., compresses the output.\n", + " - Packages the output for the the receiving application -- e.g., add headings, make JSON object, etc.\n", + "\n", + "Both the preprocessing and post-processing functions are converted to static graphs which are fused to the model. The output from the underlying model is passed to the post-processing function. The post-processing function passes the converted/packaged output back to the HTTP server. The HTTP server returns the output as the HTTP response content.\n", + "\n", + "One consideration you need to consider when building serving functions for TF.Keras models is that they run as static graphs. That means, you cannot use TF graph operations that require a dynamic graph. If you do, you will get an error during the compile of the serving function which will indicate that you are using an EagerTensor which is not supported." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_image" + }, + "source": [ + "### Serving function for image data\n", + "\n", + "To pass images to the prediction service, you encode the compressed (e.g., JPEG) image bytes into base 64 -- which makes the content safe from modification while transmitting binary data over the network. Since this deployed model expects input data as raw (uncompressed) bytes, you need to ensure that the base 64 encoded data gets converted back to raw bytes before it is passed as input to the deployed model.\n", + "\n", + "To resolve this, define a serving function (`serving_fn`) and attach it to the model as a preprocessing step. Add a `@tf.function` decorator so the serving function is fused to the underlying model (instead of upstream on a CPU).\n", + "\n", + "When you send a prediction or explanation request, the content of the request is base 64 decoded into a Tensorflow string (`tf.string`), which is passed to the serving function (`serving_fn`). The serving function preprocesses the `tf.string` into raw (uncompressed) numpy bytes (`preprocess_fn`) to match the input requirements of the model:\n", + "- `io.decode_jpeg`- Decompresses the JPG image which is returned as a Tensorflow tensor with three channels (RGB).\n", + "- `image.convert_image_dtype` - Changes integer pixel values to float 32.\n", + "- `image.resize` - Resizes the image to match the input shape for the model.\n", + "- `resized / 255.0` - Rescales (normalization) the pixel data between 0 and 1.\n", + "\n", + "At this point, the data can be passed to the model (`m_call`)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_image" + }, + "outputs": [], + "source": [ + "CONCRETE_INPUT = \"numpy_inputs\"\n", + "\n", + "\n", + "def _preprocess(bytes_input):\n", + " decoded = tf.io.decode_jpeg(bytes_input, channels=3)\n", + " decoded = tf.image.convert_image_dtype(decoded, tf.float32)\n", + " resized = tf.image.resize(decoded, size=(32, 32))\n", + " rescale = tf.cast(resized / 255.0, tf.float32)\n", + " return rescale\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def preprocess_fn(bytes_inputs):\n", + " decoded_images = tf.map_fn(\n", + " _preprocess, bytes_inputs, dtype=tf.float32, back_prop=False\n", + " )\n", + " return {\n", + " CONCRETE_INPUT: decoded_images\n", + " } # User needs to make sure the key matches model's input\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def serving_fn(bytes_inputs):\n", + " images = preprocess_fn(bytes_inputs)\n", + " prob = m_call(**images)\n", + " return prob\n", + "\n", + "\n", + "m_call = tf.function(model.call).get_concrete_function(\n", + " [tf.TensorSpec(shape=[None, 32, 32, 3], dtype=tf.float32, name=CONCRETE_INPUT)]\n", + ")\n", + "\n", + "tf.saved_model.save(\n", + " model, model_path_to_deploy, signatures={\"serving_default\": serving_fn}\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_signature:image" + }, + "source": [ + "## Get the serving function signature\n", + "\n", + "You can get the signatures of your model's input and output layers by reloading the model into memory, and querying it for the signatures corresponding to each layer.\n", + "\n", + "For your purpose, you need the signature of the serving function. Why? Well, when we send our data for prediction as a HTTP request packet, the image data is base64 encoded, and our TF.Keras model takes numpy input. Your serving function will do the conversion from base64 to a numpy array.\n", + "\n", + "When making a prediction request, you need to route the request to the serving function instead of the model, so you need to know the input layer name of the serving function -- which you will use later when you make a prediction request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_signature:image" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_path_to_deploy)\n", + "\n", + "serving_input = list(\n", + " loaded.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "print(\"Serving function input:\", serving_input)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "upload_the_model" + }, + "source": [ + "### Upload the model\n", + "\n", + "Use this helper function `upload_model` to upload your model, stored in SavedModel format, up to the `Model` service, which will instantiate a Vertex `Model` resource instance for your model. Once you've done that, you can use the `Model` resource instance in the same way as any other Vertex `Model` resource instance, such as deploying to an `Endpoint` resource for serving predictions.\n", + "\n", + "The helper function takes the following parameters:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` service.\n", + "- `image_uri`: The container image for the model deployment.\n", + "- `model_uri`: The Cloud Storage path to our SavedModel artificat. For this tutorial, this is the Cloud Storage location where the `trainer/task.py` saved the model artifacts, which we specified in the variable `MODEL_DIR`.\n", + "\n", + "The helper function calls the `Model` client service's method `upload_model`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for `Dataset`, `Model` and `Endpoint` resources.\n", + "- `model`: The specification for the Vertex `Model` resource instance.\n", + "\n", + "Let's now dive deeper into the Vertex model specification `model`. This is a dictionary object that consists of the following fields:\n", + "\n", + "- `display_name`: A human readable name for the `Model` resource.\n", + "- `metadata_schema_uri`: Since your model was built without an Vertex `Dataset` resource, you will leave this blank (`''`).\n", + "- `artificat_uri`: The Cloud Storage path where the model is stored in SavedModel format.\n", + "- `container_spec`: This is the specification for the Docker container that will be installed on the `Endpoint` resource, from which the `Model` resource will serve predictions. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + "\n", + "Uploading a model into a Vertex Model resource returns a long running operation, since it may take a few moments. You call response.result(), which is a synchronous call and will return when the Vertex Model resource is ready.\n", + "\n", + "The helper function returns the Vertex fully qualified identifier for the corresponding Vertex Model instance upload_model_response.model. You will save the identifier for subsequent steps in the variable model_to_deploy_id." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "upload_the_model" + }, + "outputs": [], + "source": [ + "IMAGE_URI = DEPLOY_IMAGE\n", + "\n", + "\n", + "def upload_model(display_name, image_uri, model_uri):\n", + " model = {\n", + " \"display_name\": display_name,\n", + " \"metadata_schema_uri\": \"\",\n", + " \"artifact_uri\": model_uri,\n", + " \"container_spec\": {\n", + " \"image_uri\": image_uri,\n", + " \"command\": [],\n", + " \"args\": [],\n", + " \"env\": [{\"name\": \"env_name\", \"value\": \"env_value\"}],\n", + " \"ports\": [{\"container_port\": 8080}],\n", + " \"predict_route\": \"\",\n", + " \"health_route\": \"\",\n", + " },\n", + " }\n", + " response = clients[\"model\"].upload_model(parent=PARENT, model=model)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " upload_model_response = response.result(timeout=180)\n", + " print(\"upload_model_response\")\n", + " print(\" model:\", upload_model_response.model)\n", + " return upload_model_response.model\n", + "\n", + "\n", + "model_to_deploy_id = upload_model(\n", + " \"cifar10-\" + TIMESTAMP, IMAGE_URI, model_path_to_deploy\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_model" + }, + "source": [ + "### Get `Model` resource information\n", + "\n", + "Now let's get the model information for just your model. Use this helper function `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource.\n", + "\n", + "This helper function calls the Vertex `Model` client service's method `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_model" + }, + "outputs": [], + "source": [ + "def get_model(name):\n", + " response = clients[\"model\"].get_model(name=name)\n", + " print(response)\n", + "\n", + "\n", + "get_model(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:custom" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex custom `Model` resource. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"cifar10_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:dedicated" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `dedicated_resources`: This refers to how many compute instances (replicas) that are scaled for serving prediction requests.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `min_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:dedicated" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"cifar10_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " \"machine_spec\": machine_spec,\n", + " },\n", + " \"disable_container_logging\": False,\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_item:test" + }, + "source": [ + "### Get test item\n", + "\n", + "You will use an example out of the test (holdout) portion of the dataset as a test item." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:test" + }, + "outputs": [], + "source": [ + "test_image = x_test[0]\n", + "test_label = y_test[0]\n", + "print(test_image.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "prepare_test_item:test,image" + }, + "source": [ + "### Prepare the request content\n", + "You are going to send the CIFAR10 image as compressed JPG image, instead of the raw uncompressed bytes:\n", + "\n", + "- `cv2.imwrite`: Use openCV to write the uncompressed image to disk as a compressed JPEG image.\n", + " - Denormalize the image data from \\[0,1) range back to [0,255).\n", + " - Convert the 32-bit floating point values to 8-bit unsigned integers.\n", + "- `tf.io.read_file`: Read the compressed JPG images back into memory as raw bytes.\n", + "- `base64.b64encode`: Encode the raw bytes into a base 64 encoded string." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "prepare_test_item:test,image" + }, + "outputs": [], + "source": [ + "import base64\n", + "\n", + "import cv2\n", + "\n", + "cv2.imwrite(\"tmp.jpg\", (test_image * 255).astype(np.uint8))\n", + "\n", + "bytes = tf.io.read_file(\"tmp.jpg\")\n", + "b64str = base64.b64encode(bytes.numpy()).decode(\"utf-8\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "send_prediction_request:image" + }, + "source": [ + "### Send the prediction request\n", + "\n", + "Ok, now you have a test image. Use this helper function `predict_image`, which takes the following parameters:\n", + "\n", + "- `image`: The test image data as a numpy array.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed to.\n", + "- `parameters_dict`: Additional parameters for serving.\n", + "\n", + "This function calls the prediction client service `predict` method with the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed to.\n", + "- `instances`: A list of instances (encoded images) to predict.\n", + "- `parameters`: Additional parameters for serving.\n", + "\n", + "To pass the image data to the prediction service, in the previous step you encoded the bytes into base64 -- which makes the content safe from modification when transmitting binary data over the network. You need to tell the serving binary where your model is deployed to, that the content has been base64 encoded, so it will decode it on the other end in the serving binary.\n", + "\n", + "Each instance in the prediction request is a dictionary entry of the form:\n", + "\n", + " {serving_input: {'b64': content}}\n", + "\n", + "- `input_name`: the name of the input layer of the underlying model.\n", + "- `'b64'`: A key that indicates the content is base64 encoded.\n", + "- `content`: The compressed JPG image bytes as a base64 encoded string.\n", + "\n", + "Since the `predict()` service can take multiple images (instances), you will send your single image as a list of one image. As a final step, you package the instances list into Google's protobuf format -- which is what we pass to the `predict()` service.\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding image in the request. You will see in the output for each prediction:\n", + "\n", + "- `predictions`: Confidence level for the prediction, between 0 and 1, for each of the classes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "send_prediction_request:image" + }, + "outputs": [], + "source": [ + "def predict_image(image, endpoint, parameters_dict):\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [{serving_input: {\"b64\": image}}]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters_dict\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", prediction)\n", + "\n", + "\n", + "predict_image(b64str, endpoint_id, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_custom_image_classification_online_pipeline.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_custom_image_classification_online_raw.ipynb b/notebooks/community/gapic/custom/showcase_custom_image_classification_online_raw.ipynb new file mode 100644 index 000000000..555137076 --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_custom_image_classification_online_raw.ipynb @@ -0,0 +1,2179 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Custom training image classification model for online prediction with raw bytes input\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom,raw" + }, + "source": [ + "## Overview\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to train and deploy a custom raw bytes model for online prediction. The purpose is to demonstrate how the serving function is constructed when the input to a serving function is raw bytes (not image, text, or tabular format). For demonstration purposes, the underlying model is an image classification model, and the image is reconstructed in the serving function." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:custom,cifar10,icn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [CIFAR10 dataset](https://www.tensorflow.org/datasets/catalog/cifar10) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use is built into TensorFlow. The trained model predicts which type of class an image is from ten classes: airplane, automobile, bird, cat, deer, dog, frog, horse, ship, truck." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,raw,online_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you will learn how to create a custom model from a Python script in a Docker container using the Vertex client library, and then do a prediction on the deployed model by sending raw byte data. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex custom job for training a model.\n", + "- Train a TensorFlow model using a custom container.\n", + "- Retrieve and load the model artifacts.\n", + "- View the model evaluation.\n", + "- Construct serving function for pre-processing raw bytes.\n", + "- Upload the model as a Vertex `Model` resource.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for training and prediction.\n", + "\n", + "Set the variables `TRAIN_GPU/TRAIN_NGPU` and `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU.\n", + "\n", + "*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3 -- which is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n", + " TRAIN_GPU, TRAIN_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n", + " )\n", + "else:\n", + " TRAIN_GPU, TRAIN_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (None, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:training,prediction" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for training and prediction\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-3:latest`\n", + " - TensorFlow 2.4\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-4:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/training/xgboost-cpu.1-1`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/training/scikit-learn-cpu.0-23:latest`\n", + " - Pytorch\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-5:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-6:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-7:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for training](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers).\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-3:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-90:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-82:latest`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-23:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-22:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-20:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for prediction](https://cloud.google.com/vertex-ai/docs/predictions/pre-built-containers)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf2-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf2-cpu.{}\".format(TF)\n", + "else:\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "TRAIN_IMAGE = \"gcr.io/cloud-aiplatform/training/{}:latest\".format(TRAIN_VERSION)\n", + "DEPLOY_IMAGE = \"gcr.io/cloud-aiplatform/prediction/{}:latest\".format(DEPLOY_VERSION)\n", + "\n", + "print(\"Training:\", TRAIN_IMAGE, TRAIN_GPU, TRAIN_NGPU)\n", + "print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU, DEPLOY_NGPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:training,prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for training and prediction.\n", + "\n", + "- Set the variables `TRAIN_COMPUTE` and `DEPLOY_COMPUTE` to configure the compute resources for the VMs you will use for for training and prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: The following is not supported for training:*\n", + "\n", + " - `standard`: 2 vCPUs\n", + " - `highcpu`: 2, 4 and 8 vCPUs\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_TRAIN_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Train machine type\", TRAIN_COMPUTE)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own custom model and training for CIFAR10." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:custom" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Model Service for `Model` resources.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:custom" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"job\"] = create_job_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_model" + }, + "source": [ + "## Train a model\n", + "\n", + "There are two ways you can train a custom model using a container image:\n", + "\n", + "- **Use a Google Cloud prebuilt container**. If you use a prebuilt container, you will additionally specify a Python package to install into the container image. This Python package contains your code for training a custom model.\n", + "\n", + "- **Use your own custom container image**. If you use your own container, the container needs to contain your code for training a custom model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_specification:prebuilt_container" + }, + "source": [ + "## Prepare your custom job specification\n", + "\n", + "Now that your clients are ready, your first step is to create a Job Specification for your custom training job. The job specification will consist of the following:\n", + "\n", + "- `worker_pool_spec` : The specification of the type of machine(s) you will use for training and how many (single or distributed)\n", + "- `python_package_spec` : The specification of the Python package to be installed with the pre-built container." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "source": [ + "### Prepare your machine specification\n", + "\n", + "Now define the machine specification for your custom training job. This tells Vertex what type of machine instance to provision for the training.\n", + " - `machine_type`: The type of GCP instance to provision -- e.g., n1-standard-8.\n", + " - `accelerator_type`: The type, if any, of hardware accelerator. In this tutorial if you previously set the variable `TRAIN_GPU != None`, you are using a GPU; otherwise you will use a CPU.\n", + " - `accelerator_count`: The number of accelerators." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "outputs": [], + "source": [ + "if TRAIN_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": TRAIN_COMPUTE,\n", + " \"accelerator_type\": TRAIN_GPU,\n", + " \"accelerator_count\": TRAIN_NGPU,\n", + " }\n", + "else:\n", + " machine_spec = {\"machine_type\": TRAIN_COMPUTE, \"accelerator_count\": 0}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "source": [ + "### Prepare your disk specification\n", + "\n", + "(optional) Now define the disk specification for your custom training job. This tells Vertex what type and size of disk to provision in each machine instance for the training.\n", + "\n", + " - `boot_disk_type`: Either SSD or Standard. SSD is faster, and Standard is less expensive. Defaults to SSD.\n", + " - `boot_disk_size_gb`: Size of disk in GB." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "outputs": [], + "source": [ + "DISK_TYPE = \"pd-ssd\" # [ pd-ssd, pd-standard]\n", + "DISK_SIZE = 200 # GB\n", + "\n", + "disk_spec = {\"boot_disk_type\": DISK_TYPE, \"boot_disk_size_gb\": DISK_SIZE}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container" + }, + "source": [ + "### Define the worker pool specification\n", + "\n", + "Next, you define the worker pool specification for your custom training job. The worker pool specification will consist of the following:\n", + "\n", + "- `replica_count`: The number of instances to provision of this machine type.\n", + "- `machine_spec`: The hardware specification.\n", + "- `disk_spec` : (optional) The disk storage specification.\n", + "\n", + "- `python_package`: The Python training package to install on the VM instance(s) and which Python module to invoke, along with command line arguments for the Python module.\n", + "\n", + "Let's dive deeper now into the python package specification:\n", + "\n", + "-`executor_image_spec`: This is the docker image which is configured for your custom training job.\n", + "\n", + "-`package_uris`: This is a list of the locations (URIs) of your python training packages to install on the provisioned instance. The locations need to be in a Cloud Storage bucket. These can be either individual python files or a zip (archive) of an entire package. In the later case, the job service will unzip (unarchive) the contents into the docker image.\n", + "\n", + "-`python_module`: The Python module (script) to invoke for running the custom training job. In this example, you will be invoking `trainer.task.py` -- note that it was not neccessary to append the `.py` suffix.\n", + "\n", + "-`args`: The command line arguments to pass to the corresponding Pythom module. In this example, you will be setting:\n", + " - `\"--model-dir=\" + MODEL_DIR` : The Cloud Storage location where to store the model artifacts. There are two ways to tell the training script where to save the model artifacts:\n", + " - direct: You pass the Cloud Storage location as a command line argument to your training script (set variable `DIRECT = True`), or\n", + " - indirect: The service passes the Cloud Storage location as the environment variable `AIP_MODEL_DIR` to your training script (set variable `DIRECT = False`). In this case, you tell the service the model artifact location in the job specification.\n", + " - `\"--epochs=\" + EPOCHS`: The number of epochs for training.\n", + " - `\"--steps=\" + STEPS`: The number of steps (batches) per epoch.\n", + " - `\"--distribute=\" + TRAIN_STRATEGY\"` : The training distribution strategy to use for single or distributed training.\n", + " - `\"single\"`: single device.\n", + " - `\"mirror\"`: all GPU devices on a single compute instance.\n", + " - `\"multi\"`: all GPU devices on all compute instances." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_job_\" + TIMESTAMP\n", + "MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, JOB_NAME)\n", + "\n", + "if not TRAIN_NGPU or TRAIN_NGPU < 2:\n", + " TRAIN_STRATEGY = \"single\"\n", + "else:\n", + " TRAIN_STRATEGY = \"mirror\"\n", + "\n", + "EPOCHS = 20\n", + "STEPS = 100\n", + "\n", + "DIRECT = True\n", + "if DIRECT:\n", + " CMDARGS = [\n", + " \"--model-dir=\" + MODEL_DIR,\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "else:\n", + " CMDARGS = [\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "\n", + "worker_pool_spec = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": machine_spec,\n", + " \"disk_spec\": disk_spec,\n", + " \"python_package_spec\": {\n", + " \"executor_image_uri\": TRAIN_IMAGE,\n", + " \"package_uris\": [BUCKET_NAME + \"/trainer_cifar10.tar.gz\"],\n", + " \"python_module\": \"trainer.task\",\n", + " \"args\": CMDARGS,\n", + " },\n", + " }\n", + "]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "assemble_custom_job_specification" + }, + "source": [ + "### Assemble a job specification\n", + "\n", + "Now assemble the complete description for the custom job specification:\n", + "\n", + "- `display_name`: The human readable name you assign to this custom job.\n", + "- `job_spec`: The specification for the custom job.\n", + " - `worker_pool_specs`: The specification for the machine VM instances.\n", + " - `base_output_directory`: This tells the service the Cloud Storage location where to save the model artifacts (when variable `DIRECT = False`). The service will then pass the location to the training script as the environment variable `AIP_MODEL_DIR`, and the path will be of the form:\n", + "\n", + " /model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "assemble_custom_job_specification" + }, + "outputs": [], + "source": [ + "if DIRECT:\n", + " job_spec = {\"worker_pool_specs\": worker_pool_spec}\n", + "else:\n", + " job_spec = {\n", + " \"worker_pool_specs\": worker_pool_spec,\n", + " \"base_output_directory\": {\"output_uri_prefix\": MODEL_DIR},\n", + " }\n", + "\n", + "custom_job = {\"display_name\": JOB_NAME, \"job_spec\": job_spec}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "examine_training_package" + }, + "source": [ + "### Examine the training package\n", + "\n", + "#### Package layout\n", + "\n", + "Before you start the training, you will look at how a Python package is assembled for a custom training job. When unarchived, the package contains the following directory/file layout.\n", + "\n", + "- PKG-INFO\n", + "- README.md\n", + "- setup.cfg\n", + "- setup.py\n", + "- trainer\n", + " - \\_\\_init\\_\\_.py\n", + " - task.py\n", + "\n", + "The files `setup.cfg` and `setup.py` are the instructions for installing the package into the operating environment of the Docker image.\n", + "\n", + "The file `trainer/task.py` is the Python script for executing the custom training job. *Note*, when we referred to it in the worker pool specification, we replace the directory slash with a dot (`trainer.task`) and dropped the file suffix (`.py`).\n", + "\n", + "#### Package Assembly\n", + "\n", + "In the following cells, you will assemble the training package." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "examine_training_package" + }, + "outputs": [], + "source": [ + "# Make folder for Python training script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\ntag_build =\\n\\ntag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'tensorflow_datasets==1.3.0',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\nName: CIFAR10 image classification\\n\\nVersion: 0.0.0\\n\\nSummary: Demostration training script\\n\\nHome-page: www.google.com\\n\\nAuthor: Google\\n\\nAuthor-email: aferlitsch@google.com\\n\\nLicense: Public\\n\\nDescription: Demo\\n\\nPlatform: Vertex\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:cifar10" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the training script task.py. We won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n", + "- Loads CIFAR10 dataset from TF Datasets (tfds).\n", + "- Builds a model using TF.Keras model API.\n", + "- Compiles the model (`compile()`).\n", + "- Sets a training distribution strategy according to the argument `args.distribute`.\n", + "- Trains the model (`fit()`) with epochs and steps according to the arguments `args.epochs` and `args.steps`\n", + "- Saves the trained model (`save(args.model_dir)`) to the specified model directory." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:cifar10" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Single, Mirror and Multi-Machine Distributed Training for CIFAR-10\n", + "\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "import argparse\n", + "import os\n", + "import sys\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv(\"AIP_MODEL_DIR\"), type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.01, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=10, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=200, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "print('DEVICES', device_lib.list_local_devices())\n", + "\n", + "# Single Machine, single compute device\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "# Single Machine, multiple compute device\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "# Multiple Machine, multiple compute device\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "# Multi-worker configuration\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "# Preparing dataset\n", + "BUFFER_SIZE = 10000\n", + "BATCH_SIZE = 64\n", + "\n", + "\n", + "def make_datasets_unbatched():\n", + "\n", + " # Scaling CIFAR10 data from (0, 255] to (0., 1.]\n", + " def scale(image, label):\n", + " image = tf.cast(image, tf.float32)\n", + " image /= 255.0\n", + " return image, label\n", + "\n", + "\n", + " datasets, info = tfds.load(name='cifar10',\n", + " with_info=True,\n", + " as_supervised=True)\n", + " return datasets['train'].map(scale).cache().shuffle(BUFFER_SIZE).repeat()\n", + "\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_cnn_model():\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu', input_shape=(32, 32, 3)),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu'),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Flatten(),\n", + " tf.keras.layers.Dense(10, activation='softmax')\n", + " ])\n", + " model.compile(\n", + " loss=tf.keras.losses.sparse_categorical_crossentropy,\n", + " optimizer=tf.keras.optimizers.SGD(learning_rate=args.lr),\n", + " metrics=['accuracy'])\n", + " return model\n", + "\n", + "\n", + "# Train the model\n", + "NUM_WORKERS = strategy.num_replicas_in_sync\n", + "# Here the batch size scales up by number of workers since\n", + "# `tf.data.Dataset.batch` expects the global batch size.\n", + "GLOBAL_BATCH_SIZE = BATCH_SIZE * NUM_WORKERS\n", + "train_dataset = make_datasets_unbatched().batch(GLOBAL_BATCH_SIZE)\n", + "\n", + "with strategy.scope():\n", + " # Creation of dataset, and model building/compiling need to be within\n", + " # `strategy.scope()`.\n", + " model = build_and_compile_cnn_model()\n", + "\n", + "model.fit(x=train_dataset, epochs=args.epochs, steps_per_epoch=args.steps)\n", + "model.save(args.model_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tarball_training_script" + }, + "source": [ + "#### Store training script on your Cloud Storage bucket\n", + "\n", + "Next, you package the training folder into a compressed tar ball, and then store it in your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tarball_training_script" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_cifar10.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job" + }, + "source": [ + "### Train the model\n", + "\n", + "\n", + "Now start the training of your custom training job on Vertex. Use this helper function `create_custom_job`, which takes the following parameter:\n", + "\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "The helper function calls job client service's `create_custom_job` method, with the following parameters:\n", + "\n", + "-`parent`: The Vertex location path to `Dataset`, `Model` and `Endpoint` resources.\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "You will display a handful of the fields returned in `response` object, with the two that are of most interest are:\n", + "\n", + "`response.name`: The Vertex fully qualified identifier assigned to this custom training job. You save this identifier for using in subsequent steps.\n", + "\n", + "`response.state`: The current state of the custom training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job" + }, + "outputs": [], + "source": [ + "def create_custom_job(custom_job):\n", + " response = clients[\"job\"].create_custom_job(parent=PARENT, custom_job=custom_job)\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = create_custom_job(custom_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "job_id:response" + }, + "source": [ + "Now get the unique identifier for the custom job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom job\n", + "job_id = response.name\n", + "# The short numeric ID for the custom job\n", + "job_short_id = job_id.split(\"/\")[-1]\n", + "\n", + "print(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_custom_job" + }, + "source": [ + "### Get information on a custom job\n", + "\n", + "Next, use this helper function `get_custom_job`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "The helper function calls the job client service's`get_custom_job` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "If you recall, you got the Vertex fully qualified identifier for the custom job in the `response.name` field when you called the `create_custom_job` method, and saved the identifier in the variable `job_id`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_custom_job" + }, + "outputs": [], + "source": [ + "def get_custom_job(name, silent=False):\n", + " response = clients[\"job\"].get_custom_job(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = get_custom_job(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete:custom" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, we will need to know the location of the saved model, which the Python script saved in your local Cloud Storage bucket at `MODEL_DIR + '/saved_model.pb'`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete:custom" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_custom_job(job_id, True)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_path_to_deploy = None\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " if not DIRECT:\n", + " MODEL_DIR = MODEL_DIR + \"/model\"\n", + " model_path_to_deploy = MODEL_DIR\n", + " print(\"Training Time:\", response.update_time - response.create_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model_to_deploy:\", model_path_to_deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "load_saved_model" + }, + "source": [ + "## Load the saved model\n", + "\n", + "Your model is stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Now load it from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n", + "\n", + "To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "load_saved_model" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:image" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load the CIFAR10 test (holdout) data from `tf.keras.datasets`, using the method `load_data()`. This will return the dataset as a tuple of two elements. The first element is the training data and the second is the test data. Each element is also a tuple of two elements: the image data, and the corresponding labels.\n", + "\n", + "You don't need the training data, and hence why we loaded it as `(_, _)`.\n", + "\n", + "Before you can run the data through evaluation, you need to preprocess it:\n", + "\n", + "x_test:\n", + "1. Normalize (rescaling) the pixel data by dividing each pixel by 255. This will replace each single byte integer pixel with a 32-bit floating point number between 0 and 1.\n", + "\n", + "y_test:
\n", + "2. The labels are currently scalar (sparse). If you look back at the `compile()` step in the `trainer/task.py` script, you will find that it was compiled for sparse labels. So we don't need to do anything more." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "evaluate_custom_model:image,cifar10" + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "from tensorflow.keras.datasets import cifar10\n", + "\n", + "(_, _), (x_test, y_test) = cifar10.load_data()\n", + "x_test = (x_test / 255.0).astype(np.float32)\n", + "\n", + "print(x_test.shape, y_test.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "perform_evaluation_custom" + }, + "source": [ + "### Perform the model evaluation\n", + "\n", + "Now evaluate how well the model in the custom job did." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "perform_evaluation_custom" + }, + "outputs": [], + "source": [ + "model.evaluate(x_test, y_test)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "how_serving_function_works" + }, + "source": [ + "## Upload the model for serving\n", + "\n", + "Next, you will upload your TF.Keras model from the custom job to Vertex `Model` service, which will create a Vertex `Model` resource for your custom model. During upload, you need to define a serving function to convert data to the format your model expects. If you send encoded data to Vertex, your serving function ensures that the data is decoded on the model server before it is passed as input to your model.\n", + "\n", + "### How does the serving function work\n", + "\n", + "When you send a request to an online prediction server, the request is received by a HTTP server. The HTTP server extracts the prediction request from the HTTP request content body. The extracted prediction request is forwarded to the serving function. For Google pre-built prediction containers, the request content is passed to the serving function as a `tf.string`.\n", + "\n", + "The serving function consists of two parts:\n", + "\n", + "- `preprocessing function`:\n", + " - Converts the input (`tf.string`) to the input shape and data type of the underlying model (dynamic graph).\n", + " - Performs the same preprocessing of the data that was done during training the underlying model -- e.g., normalizing, scaling, etc.\n", + "- `post-processing function`:\n", + " - Converts the model output to format expected by the receiving application -- e.q., compresses the output.\n", + " - Packages the output for the the receiving application -- e.g., add headings, make JSON object, etc.\n", + "\n", + "Both the preprocessing and post-processing functions are converted to static graphs which are fused to the model. The output from the underlying model is passed to the post-processing function. The post-processing function passes the converted/packaged output back to the HTTP server. The HTTP server returns the output as the HTTP response content.\n", + "\n", + "One consideration you need to consider when building serving functions for TF.Keras models is that they run as static graphs. That means, you cannot use TF graph operations that require a dynamic graph. If you do, you will get an error during the compile of the serving function which will indicate that you are using an EagerTensor which is not supported." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_image:raw" + }, + "source": [ + "### Serving function for raw bytes\n", + "\n", + "To pass raw bytes to the prediction service, you encode the bytes into base64 -- which makes the content safe from modification while transmitting binary data over the network. Since this deployed model expects input data as raw bytes, you need to ensure that the base64 encoded data gets converted back to raw bytes before it is passed as input to the deployed model.\n", + "\n", + "To resolve this, define a serving function (`serving_fn`) and attach it to the model as a preprocessing step. Add a `@tf.function` decorator so the serving function is fused to the underlying model (instead of upstream on a CPU).\n", + "\n", + "When you send a prediction or explanation request, the content of the request is base 64 decoded into a TensorFlow string, which is passed to the serving function (`serving_fn`). The serving function preprocesses the tf.string into raw numpy bytes (`preprocess_fn`) to match the input requirements of the model:\n", + "\n", + "- `tf.io.decode_raw`: Decodes base64 encoded tf.string into a decoded byte tensor.\n", + "- `tf.reshape`: Reshapes the byte tensor to (32, 32, 3) -- i.e., CIFAR10.\n", + "- `tf.cast`: Converts the integer pixel values to floating point.\n", + "- `recasts / 255.0`: Rescales (normalization) the pixel data between 0 and 1.\n", + "\n", + "At this point, the data can be passed to the model (`m_call`)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_image:raw" + }, + "outputs": [], + "source": [ + "CONCRETE_INPUT = \"numpy_inputs\"\n", + "\n", + "\n", + "def _preprocess(bytes_input):\n", + " decoded = tf.io.decode_raw(bytes_input, tf.uint8)\n", + " resized = tf.reshape(decoded, shape=(32, 32, 3))\n", + " recast = tf.cast(resized, tf.float32)\n", + " rescale = tf.cast(recast / 255.0, tf.float32)\n", + " return rescale\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def preprocess_fn(bytes_inputs):\n", + " decoded_images = tf.map_fn(\n", + " _preprocess, bytes_inputs, dtype=tf.float32, back_prop=False\n", + " )\n", + " return {\n", + " CONCRETE_INPUT: decoded_images\n", + " } # User needs to make sure the key matches model's input\n", + "\n", + "\n", + "m_call = tf.function(model.call).get_concrete_function(\n", + " [tf.TensorSpec(shape=[None, 32, 32, 3], dtype=tf.float32, name=CONCRETE_INPUT)]\n", + ")\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def serving_fn(bytes_inputs):\n", + " images = preprocess_fn(bytes_inputs)\n", + " prob = m_call(**images)\n", + " return prob\n", + "\n", + "\n", + "tf.saved_model.save(\n", + " model,\n", + " model_path_to_deploy,\n", + " signatures={\n", + " \"serving_default\": serving_fn,\n", + " },\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_signature:image" + }, + "source": [ + "## Get the serving function signature\n", + "\n", + "You can get the signatures of your model's input and output layers by reloading the model into memory, and querying it for the signatures corresponding to each layer.\n", + "\n", + "For your purpose, you need the signature of the serving function. Why? Well, when we send our data for prediction as a HTTP request packet, the image data is base64 encoded, and our TF.Keras model takes numpy input. Your serving function will do the conversion from base64 to a numpy array.\n", + "\n", + "When making a prediction request, you need to route the request to the serving function instead of the model, so you need to know the input layer name of the serving function -- which you will use later when you make a prediction request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_signature:image" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_path_to_deploy)\n", + "\n", + "serving_input = list(\n", + " loaded.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "print(\"Serving function input:\", serving_input)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "upload_the_model" + }, + "source": [ + "### Upload the model\n", + "\n", + "Use this helper function `upload_model` to upload your model, stored in SavedModel format, up to the `Model` service, which will instantiate a Vertex `Model` resource instance for your model. Once you've done that, you can use the `Model` resource instance in the same way as any other Vertex `Model` resource instance, such as deploying to an `Endpoint` resource for serving predictions.\n", + "\n", + "The helper function takes the following parameters:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` service.\n", + "- `image_uri`: The container image for the model deployment.\n", + "- `model_uri`: The Cloud Storage path to our SavedModel artificat. For this tutorial, this is the Cloud Storage location where the `trainer/task.py` saved the model artifacts, which we specified in the variable `MODEL_DIR`.\n", + "\n", + "The helper function calls the `Model` client service's method `upload_model`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for `Dataset`, `Model` and `Endpoint` resources.\n", + "- `model`: The specification for the Vertex `Model` resource instance.\n", + "\n", + "Let's now dive deeper into the Vertex model specification `model`. This is a dictionary object that consists of the following fields:\n", + "\n", + "- `display_name`: A human readable name for the `Model` resource.\n", + "- `metadata_schema_uri`: Since your model was built without an Vertex `Dataset` resource, you will leave this blank (`''`).\n", + "- `artificat_uri`: The Cloud Storage path where the model is stored in SavedModel format.\n", + "- `container_spec`: This is the specification for the Docker container that will be installed on the `Endpoint` resource, from which the `Model` resource will serve predictions. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + "\n", + "Uploading a model into a Vertex Model resource returns a long running operation, since it may take a few moments. You call response.result(), which is a synchronous call and will return when the Vertex Model resource is ready.\n", + "\n", + "The helper function returns the Vertex fully qualified identifier for the corresponding Vertex Model instance upload_model_response.model. You will save the identifier for subsequent steps in the variable model_to_deploy_id." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "upload_the_model" + }, + "outputs": [], + "source": [ + "IMAGE_URI = DEPLOY_IMAGE\n", + "\n", + "\n", + "def upload_model(display_name, image_uri, model_uri):\n", + " model = {\n", + " \"display_name\": display_name,\n", + " \"metadata_schema_uri\": \"\",\n", + " \"artifact_uri\": model_uri,\n", + " \"container_spec\": {\n", + " \"image_uri\": image_uri,\n", + " \"command\": [],\n", + " \"args\": [],\n", + " \"env\": [{\"name\": \"env_name\", \"value\": \"env_value\"}],\n", + " \"ports\": [{\"container_port\": 8080}],\n", + " \"predict_route\": \"\",\n", + " \"health_route\": \"\",\n", + " },\n", + " }\n", + " response = clients[\"model\"].upload_model(parent=PARENT, model=model)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " upload_model_response = response.result(timeout=180)\n", + " print(\"upload_model_response\")\n", + " print(\" model:\", upload_model_response.model)\n", + " return upload_model_response.model\n", + "\n", + "\n", + "model_to_deploy_id = upload_model(\n", + " \"cifar10-\" + TIMESTAMP, IMAGE_URI, model_path_to_deploy\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_model" + }, + "source": [ + "### Get `Model` resource information\n", + "\n", + "Now let's get the model information for just your model. Use this helper function `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource.\n", + "\n", + "This helper function calls the Vertex `Model` client service's method `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_model" + }, + "outputs": [], + "source": [ + "def get_model(name):\n", + " response = clients[\"model\"].get_model(name=name)\n", + " print(response)\n", + "\n", + "\n", + "get_model(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:custom" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex custom `Model` resource. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"cifar10_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:dedicated" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `dedicated_resources`: This refers to how many compute instances (replicas) that are scaled for serving prediction requests.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `min_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:dedicated" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"cifar10_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " \"machine_spec\": machine_spec,\n", + " },\n", + " \"disable_container_logging\": False,\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_item:test" + }, + "source": [ + "### Get test item\n", + "\n", + "You will use an example out of the test (holdout) portion of the dataset as a test item." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:test" + }, + "outputs": [], + "source": [ + "test_image = x_test[0]\n", + "test_label = y_test[0]\n", + "print(test_image.shape)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "prepare_test_item:raw" + }, + "outputs": [], + "source": [ + "import base64\n", + "\n", + "bytes = (test_image * 255).astype(np.uint8).tobytes()\n", + "b64str = base64.b64encode(np.ascontiguousarray(bytes)).decode(\"utf-8\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "send_prediction_request:image" + }, + "source": [ + "### Send the prediction request\n", + "\n", + "Ok, now you have a test image. Use this helper function `predict_image`, which takes the following parameters:\n", + "\n", + "- `image`: The test image data as a numpy array.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed to.\n", + "- `parameters_dict`: Additional parameters for serving.\n", + "\n", + "This function calls the prediction client service `predict` method with the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed to.\n", + "- `instances`: A list of instances (encoded images) to predict.\n", + "- `parameters`: Additional parameters for serving.\n", + "\n", + "To pass the image data to the prediction service, in the previous step you encoded the bytes into base64 -- which makes the content safe from modification when transmitting binary data over the network. You need to tell the serving binary where your model is deployed to, that the content has been base64 encoded, so it will decode it on the other end in the serving binary.\n", + "\n", + "Each instance in the prediction request is a dictionary entry of the form:\n", + "\n", + " {serving_input: {'b64': content}}\n", + "\n", + "- `input_name`: the name of the input layer of the underlying model.\n", + "- `'b64'`: A key that indicates the content is base64 encoded.\n", + "- `content`: The compressed JPG image bytes as a base64 encoded string.\n", + "\n", + "Since the `predict()` service can take multiple images (instances), you will send your single image as a list of one image. As a final step, you package the instances list into Google's protobuf format -- which is what we pass to the `predict()` service.\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding image in the request. You will see in the output for each prediction:\n", + "\n", + "- `predictions`: Confidence level for the prediction, between 0 and 1, for each of the classes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "send_prediction_request:image" + }, + "outputs": [], + "source": [ + "def predict_image(image, endpoint, parameters_dict):\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [{serving_input: {\"b64\": image}}]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters_dict\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", prediction)\n", + "\n", + "\n", + "predict_image(b64str, endpoint_id, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_custom_image_classification_online_raw.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_custom_image_classification_online_tfserving.ipynb b/notebooks/community/gapic/custom/showcase_custom_image_classification_online_tfserving.ipynb new file mode 100644 index 000000000..0603826bf --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_custom_image_classification_online_tfserving.ipynb @@ -0,0 +1,2269 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Custom training image classification model using TF Serving container for online prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom,training,tfserving" + }, + "source": [ + "## Overview\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to train and deploy a custom image classification model for online prediction, using TF Serving container for prediction." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:custom,cifar10,icn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [CIFAR10 dataset](https://www.tensorflow.org/datasets/catalog/cifar10) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use is built into TensorFlow. The trained model predicts which type of class an image is from ten classes: airplane, automobile, bird, cat, deer, dog, frog, horse, ship, truck." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,training,online_prediction,tfserving" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial you create a custom model from a Python script in a Google prebuilt Docker container using the Vertex client library, and then do a prediction on the deployed model using TF Serving container. You can alternatively create custom models using `gcloud` command-line tool or online using Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex custom job for training a model.\n", + "- Create a `TrainingPipeline` resource.\n", + "- Train a TensorFlow model.\n", + "- Retrieve and load the model artifacts.\n", + "- View the model evaluation.\n", + "- Upload the model as a Vertex `Model` resource.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource with a TF Serving container.\n", + "- Make a prediction.\n", + "- Undeploy the `Model` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_docker" + }, + "source": [ + "### Install Docker (Colab or Local)\n", + "\n", + "If you are using Google Cloud Notebook, Docker is already installed. Skip these steps.\n", + "\n", + "By default, Docker is not installed on Colab. If you're running colab, then you need to do the following to install Docker." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_docker" + }, + "outputs": [], + "source": [ + "if \"google.colab\" in sys.modules:\n", + " ! sudo apt update\n", + " ! sudo apt install apt-transport-https ca-certificates curl software-properties-common\n", + " ! curl -fsSL https://download.docker.com/linux/ubuntu/gpg | sudo apt-key add -\n", + " ! sudo add-apt-repository \"deb [arch=amd64] https://download.docker.com/linux/ubuntu bionic stable\"\n", + " ! sudo apt update\n", + " ! sudo apt install docker-ce" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "start_docker_service" + }, + "source": [ + "#### Start Docker service\n", + "\n", + "Start the `docker` service." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "start_docker_service" + }, + "outputs": [], + "source": [ + "if \"google.colab\" in sys.modules:\n", + " ! sudo service docker start" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for training and prediction.\n", + "\n", + "Set the variables `TRAIN_GPU/TRAIN_NGPU` and `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU.\n", + "\n", + "*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3 -- which is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n", + " TRAIN_GPU, TRAIN_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n", + " )\n", + "else:\n", + " TRAIN_GPU, TRAIN_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (None, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:training" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for training.\n", + "\n", + "- Set the variable `TF` to the TensorFlow version of the container image. For example, `2-1` would be version 2.1, and `1-15` would be version 1.15. The following list shows some of the pre-built images available:\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-3:latest`\n", + " - TensorFlow 2.4\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-4:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/training/xgboost-cpu.1-1`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/training/scikit-learn-cpu.0-23:latest`\n", + " - Pytorch\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-5:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-6:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-7:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for training](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:training" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + "else:\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "TRAIN_IMAGE = \"gcr.io/cloud-aiplatform/training/{}:latest\".format(TRAIN_VERSION)\n", + "\n", + "print(\"Training:\", TRAIN_IMAGE, TRAIN_GPU, TRAIN_NGPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:tfserving" + }, + "source": [ + "#### Container (Docker) image for prediction\n", + "\n", + "Next, you set the TF Serving Docker container image for prediction.\n", + "\n", + " 1. Pull the corresponding CPU or GPU Docker image for TF Serving from Docker Hub.\n", + " 2. Create a tag for registering the image with Cloud Container Registry (gcr.io)\n", + " 3. Register the image with Cloud Container Registry.\n", + "\n", + "For more details on using TF Serving container, see [TF Serving](https://www.tensorflow.org/tfx/serving/docker)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:tfserving" + }, + "outputs": [], + "source": [ + "if DEPLOY_GPU:\n", + " ! docker pull tensorflow/serving:latest-gpu\n", + " DEPLOY_IMAGE = \"gcr.io/\" + PROJECT_ID + \"/tf_serving:gpu\"\n", + "else:\n", + " ! docker pull tensorflow/serving:latest\n", + " DEPLOY_IMAGE = \"gcr.io/\" + PROJECT_ID + \"/tf_serving\"\n", + "\n", + "! docker tag tensorflow/serving $DEPLOY_IMAGE\n", + "! docker push $DEPLOY_IMAGE\n", + "\n", + "print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU, DEPLOY_NGPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:training,prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for training and prediction.\n", + "\n", + "- Set the variables `TRAIN_COMPUTE` and `DEPLOY_COMPUTE` to configure the compute resources for the VMs you will use for for training and prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: The following is not supported for training:*\n", + "\n", + " - `standard`: 2 vCPUs\n", + " - `highcpu`: 2, 4 and 8 vCPUs\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_TRAIN_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Train machine type\", TRAIN_COMPUTE)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own custom model and training for CIFAR10." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:custom" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Model Service for `Model` resources.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:custom" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"job\"] = create_job_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_model" + }, + "source": [ + "## Train a model\n", + "\n", + "There are two ways you can train a custom model using a container image:\n", + "\n", + "- **Use a Google Cloud prebuilt container**. If you use a prebuilt container, you will additionally specify a Python package to install into the container image. This Python package contains your code for training a custom model.\n", + "\n", + "- **Use your own custom container image**. If you use your own container, the container needs to contain your code for training a custom model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_specification:prebuilt_container" + }, + "source": [ + "## Prepare your custom job specification\n", + "\n", + "Now that your clients are ready, your first step is to create a Job Specification for your custom training job. The job specification will consist of the following:\n", + "\n", + "- `worker_pool_spec` : The specification of the type of machine(s) you will use for training and how many (single or distributed)\n", + "- `python_package_spec` : The specification of the Python package to be installed with the pre-built container." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "source": [ + "### Prepare your machine specification\n", + "\n", + "Now define the machine specification for your custom training job. This tells Vertex what type of machine instance to provision for the training.\n", + " - `machine_type`: The type of GCP instance to provision -- e.g., n1-standard-8.\n", + " - `accelerator_type`: The type, if any, of hardware accelerator. In this tutorial if you previously set the variable `TRAIN_GPU != None`, you are using a GPU; otherwise you will use a CPU.\n", + " - `accelerator_count`: The number of accelerators." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "outputs": [], + "source": [ + "if TRAIN_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": TRAIN_COMPUTE,\n", + " \"accelerator_type\": TRAIN_GPU,\n", + " \"accelerator_count\": TRAIN_NGPU,\n", + " }\n", + "else:\n", + " machine_spec = {\"machine_type\": TRAIN_COMPUTE, \"accelerator_count\": 0}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "source": [ + "### Prepare your disk specification\n", + "\n", + "(optional) Now define the disk specification for your custom training job. This tells Vertex what type and size of disk to provision in each machine instance for the training.\n", + "\n", + " - `boot_disk_type`: Either SSD or Standard. SSD is faster, and Standard is less expensive. Defaults to SSD.\n", + " - `boot_disk_size_gb`: Size of disk in GB." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "outputs": [], + "source": [ + "DISK_TYPE = \"pd-ssd\" # [ pd-ssd, pd-standard]\n", + "DISK_SIZE = 200 # GB\n", + "\n", + "disk_spec = {\"boot_disk_type\": DISK_TYPE, \"boot_disk_size_gb\": DISK_SIZE}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container,tfserving" + }, + "source": [ + "### Define the worker pool specification\n", + "\n", + "Next, you define the worker pool specification for your custom training job. The worker pool specification will consist of the following:\n", + "\n", + "- `replica_count`: The number of instances to provision of this machine type.\n", + "- `machine_spec`: The hardware specification.\n", + "- `disk_spec` : (optional) The disk storage specification.\n", + "\n", + "- `python_package`: The Python training package to install on the VM instance(s) and which Python module to invoke, along with command line arguments for the Python module.\n", + "\n", + "Let's dive deeper now into the python package specification:\n", + "\n", + "-`executor_image_spec`: This is the docker image which is configured for your custom training job.\n", + "\n", + "-`package_uris`: This is a list of the locations (URIs) of your python training packages to install on the provisioned instance. The locations need to be in a Cloud Storage bucket. These can be either individual python files or a zip (archive) of an entire package. In the later case, the job service will unzip (unarchive) the contents into the docker image.\n", + "\n", + "-`python_module`: The Python module (script) to invoke for running the custom training job. In this example, you will be invoking `trainer.task.py` -- note that it was not neccessary to append the `.py` suffix.\n", + "\n", + "-`args`: The command line arguments to pass to the corresponding Pythom module. In this example, you will be setting:\n", + " - `\"--model-dir=\" + MODEL_DIR` : The Cloud Storage location where to store the model artifacts. There are two ways to tell the training script where to save the model artifacts:\n", + " - direct: You pass the Cloud Storage location as a command line argument to your training script (set variable `DIRECT = True`), or\n", + " - indirect: The service passes the Cloud Storage location as the environment variable `AIP_MODEL_DIR` to your training script (set variable `DIRECT = False`). In this case, you tell the service the model artifact location in the job specification.\n", + " - `\"--epochs=\" + EPOCHS`: The number of epochs for training.\n", + " - `\"--steps=\" + STEPS`: The number of steps (batches) per epoch.\n", + " - `\"--distribute=\" + TRAIN_STRATEGY\"` : The training distribution strategy to use for single or distributed training.\n", + " - `\"single\"`: single device.\n", + " - `\"mirror\"`: all GPU devices on a single compute instance.\n", + " - `\"multi\"`: all GPU devices on all compute instances.\n", + "\n", + "*Note*: For TF Serving, the `MODEL_DIR` must end in a subfolder that is a number, e.g., 1." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container,tfserving" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_job_\" + TIMESTAMP\n", + "MODEL_DIR = \"{}/{}/1\".format(BUCKET_NAME, JOB_NAME)\n", + "\n", + "if not TRAIN_NGPU or TRAIN_NGPU < 2:\n", + " TRAIN_STRATEGY = \"single\"\n", + "else:\n", + " TRAIN_STRATEGY = \"mirror\"\n", + "\n", + "EPOCHS = 20\n", + "STEPS = 100\n", + "\n", + "DIRECT = True\n", + "if DIRECT:\n", + " CMDARGS = [\n", + " \"--model-dir=\" + MODEL_DIR,\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "else:\n", + " CMDARGS = [\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "\n", + "worker_pool_spec = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": machine_spec,\n", + " \"disk_spec\": disk_spec,\n", + " \"python_package_spec\": {\n", + " \"executor_image_uri\": TRAIN_IMAGE,\n", + " \"package_uris\": [BUCKET_NAME + \"/trainer_cifar10.tar.gz\"],\n", + " \"python_module\": \"trainer.task\",\n", + " \"args\": CMDARGS,\n", + " },\n", + " }\n", + "]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "assemble_custom_job_specification" + }, + "source": [ + "### Assemble a job specification\n", + "\n", + "Now assemble the complete description for the custom job specification:\n", + "\n", + "- `display_name`: The human readable name you assign to this custom job.\n", + "- `job_spec`: The specification for the custom job.\n", + " - `worker_pool_specs`: The specification for the machine VM instances.\n", + " - `base_output_directory`: This tells the service the Cloud Storage location where to save the model artifacts (when variable `DIRECT = False`). The service will then pass the location to the training script as the environment variable `AIP_MODEL_DIR`, and the path will be of the form:\n", + "\n", + " /model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "assemble_custom_job_specification" + }, + "outputs": [], + "source": [ + "if DIRECT:\n", + " job_spec = {\"worker_pool_specs\": worker_pool_spec}\n", + "else:\n", + " job_spec = {\n", + " \"worker_pool_specs\": worker_pool_spec,\n", + " \"base_output_directory\": {\"output_uri_prefix\": MODEL_DIR},\n", + " }\n", + "\n", + "custom_job = {\"display_name\": JOB_NAME, \"job_spec\": job_spec}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "examine_training_package" + }, + "source": [ + "### Examine the training package\n", + "\n", + "#### Package layout\n", + "\n", + "Before you start the training, you will look at how a Python package is assembled for a custom training job. When unarchived, the package contains the following directory/file layout.\n", + "\n", + "- PKG-INFO\n", + "- README.md\n", + "- setup.cfg\n", + "- setup.py\n", + "- trainer\n", + " - \\_\\_init\\_\\_.py\n", + " - task.py\n", + "\n", + "The files `setup.cfg` and `setup.py` are the instructions for installing the package into the operating environment of the Docker image.\n", + "\n", + "The file `trainer/task.py` is the Python script for executing the custom training job. *Note*, when we referred to it in the worker pool specification, we replace the directory slash with a dot (`trainer.task`) and dropped the file suffix (`.py`).\n", + "\n", + "#### Package Assembly\n", + "\n", + "In the following cells, you will assemble the training package." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "examine_training_package" + }, + "outputs": [], + "source": [ + "# Make folder for Python training script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\ntag_build =\\n\\ntag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'tensorflow_datasets==1.3.0',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\nName: CIFAR10 image classification\\n\\nVersion: 0.0.0\\n\\nSummary: Demostration training script\\n\\nHome-page: www.google.com\\n\\nAuthor: Google\\n\\nAuthor-email: aferlitsch@google.com\\n\\nLicense: Public\\n\\nDescription: Demo\\n\\nPlatform: Vertex\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:cifar10" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the training script task.py. We won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n", + "- Loads CIFAR10 dataset from TF Datasets (tfds).\n", + "- Builds a model using TF.Keras model API.\n", + "- Compiles the model (`compile()`).\n", + "- Sets a training distribution strategy according to the argument `args.distribute`.\n", + "- Trains the model (`fit()`) with epochs and steps according to the arguments `args.epochs` and `args.steps`\n", + "- Saves the trained model (`save(args.model_dir)`) to the specified model directory." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:cifar10" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Single, Mirror and Multi-Machine Distributed Training for CIFAR-10\n", + "\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "import argparse\n", + "import os\n", + "import sys\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv(\"AIP_MODEL_DIR\"), type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.01, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=10, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=200, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "print('DEVICES', device_lib.list_local_devices())\n", + "\n", + "# Single Machine, single compute device\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "# Single Machine, multiple compute device\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "# Multiple Machine, multiple compute device\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "# Multi-worker configuration\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "# Preparing dataset\n", + "BUFFER_SIZE = 10000\n", + "BATCH_SIZE = 64\n", + "\n", + "\n", + "def make_datasets_unbatched():\n", + "\n", + " # Scaling CIFAR10 data from (0, 255] to (0., 1.]\n", + " def scale(image, label):\n", + " image = tf.cast(image, tf.float32)\n", + " image /= 255.0\n", + " return image, label\n", + "\n", + "\n", + " datasets, info = tfds.load(name='cifar10',\n", + " with_info=True,\n", + " as_supervised=True)\n", + " return datasets['train'].map(scale).cache().shuffle(BUFFER_SIZE).repeat()\n", + "\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_cnn_model():\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu', input_shape=(32, 32, 3)),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu'),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Flatten(),\n", + " tf.keras.layers.Dense(10, activation='softmax')\n", + " ])\n", + " model.compile(\n", + " loss=tf.keras.losses.sparse_categorical_crossentropy,\n", + " optimizer=tf.keras.optimizers.SGD(learning_rate=args.lr),\n", + " metrics=['accuracy'])\n", + " return model\n", + "\n", + "\n", + "# Train the model\n", + "NUM_WORKERS = strategy.num_replicas_in_sync\n", + "# Here the batch size scales up by number of workers since\n", + "# `tf.data.Dataset.batch` expects the global batch size.\n", + "GLOBAL_BATCH_SIZE = BATCH_SIZE * NUM_WORKERS\n", + "train_dataset = make_datasets_unbatched().batch(GLOBAL_BATCH_SIZE)\n", + "\n", + "with strategy.scope():\n", + " # Creation of dataset, and model building/compiling need to be within\n", + " # `strategy.scope()`.\n", + " model = build_and_compile_cnn_model()\n", + "\n", + "model.fit(x=train_dataset, epochs=args.epochs, steps_per_epoch=args.steps)\n", + "model.save(args.model_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tarball_training_script" + }, + "source": [ + "#### Store training script on your Cloud Storage bucket\n", + "\n", + "Next, you package the training folder into a compressed tar ball, and then store it in your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tarball_training_script" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_cifar10.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job" + }, + "source": [ + "### Train the model\n", + "\n", + "\n", + "Now start the training of your custom training job on Vertex. Use this helper function `create_custom_job`, which takes the following parameter:\n", + "\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "The helper function calls job client service's `create_custom_job` method, with the following parameters:\n", + "\n", + "-`parent`: The Vertex location path to `Dataset`, `Model` and `Endpoint` resources.\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "You will display a handful of the fields returned in `response` object, with the two that are of most interest are:\n", + "\n", + "`response.name`: The Vertex fully qualified identifier assigned to this custom training job. You save this identifier for using in subsequent steps.\n", + "\n", + "`response.state`: The current state of the custom training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job" + }, + "outputs": [], + "source": [ + "def create_custom_job(custom_job):\n", + " response = clients[\"job\"].create_custom_job(parent=PARENT, custom_job=custom_job)\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = create_custom_job(custom_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "job_id:response" + }, + "source": [ + "Now get the unique identifier for the custom job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom job\n", + "job_id = response.name\n", + "# The short numeric ID for the custom job\n", + "job_short_id = job_id.split(\"/\")[-1]\n", + "\n", + "print(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_custom_job" + }, + "source": [ + "### Get information on a custom job\n", + "\n", + "Next, use this helper function `get_custom_job`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "The helper function calls the job client service's`get_custom_job` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "If you recall, you got the Vertex fully qualified identifier for the custom job in the `response.name` field when you called the `create_custom_job` method, and saved the identifier in the variable `job_id`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_custom_job" + }, + "outputs": [], + "source": [ + "def get_custom_job(name, silent=False):\n", + " response = clients[\"job\"].get_custom_job(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = get_custom_job(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete:custom" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, we will need to know the location of the saved model, which the Python script saved in your local Cloud Storage bucket at `MODEL_DIR + '/saved_model.pb'`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete:custom" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_custom_job(job_id, True)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_path_to_deploy = None\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " if not DIRECT:\n", + " MODEL_DIR = MODEL_DIR + \"/model\"\n", + " model_path_to_deploy = MODEL_DIR\n", + " print(\"Training Time:\", response.update_time - response.create_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model_to_deploy:\", model_path_to_deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "load_saved_model" + }, + "source": [ + "## Load the saved model\n", + "\n", + "Your model is stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Now load it from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n", + "\n", + "To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "load_saved_model" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:image" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load the CIFAR10 test (holdout) data from `tf.keras.datasets`, using the method `load_data()`. This will return the dataset as a tuple of two elements. The first element is the training data and the second is the test data. Each element is also a tuple of two elements: the image data, and the corresponding labels.\n", + "\n", + "You don't need the training data, and hence why we loaded it as `(_, _)`.\n", + "\n", + "Before you can run the data through evaluation, you need to preprocess it:\n", + "\n", + "x_test:\n", + "1. Normalize (rescaling) the pixel data by dividing each pixel by 255. This will replace each single byte integer pixel with a 32-bit floating point number between 0 and 1.\n", + "\n", + "y_test:
\n", + "2. The labels are currently scalar (sparse). If you look back at the `compile()` step in the `trainer/task.py` script, you will find that it was compiled for sparse labels. So we don't need to do anything more." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "evaluate_custom_model:image,cifar10" + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "from tensorflow.keras.datasets import cifar10\n", + "\n", + "(_, _), (x_test, y_test) = cifar10.load_data()\n", + "x_test = (x_test / 255.0).astype(np.float32)\n", + "\n", + "print(x_test.shape, y_test.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "perform_evaluation_custom" + }, + "source": [ + "### Perform the model evaluation\n", + "\n", + "Now evaluate how well the model in the custom job did." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "perform_evaluation_custom" + }, + "outputs": [], + "source": [ + "model.evaluate(x_test, y_test)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "how_serving_function_works" + }, + "source": [ + "## Upload the model for serving\n", + "\n", + "Next, you will upload your TF.Keras model from the custom job to Vertex `Model` service, which will create a Vertex `Model` resource for your custom model. During upload, you need to define a serving function to convert data to the format your model expects. If you send encoded data to Vertex, your serving function ensures that the data is decoded on the model server before it is passed as input to your model.\n", + "\n", + "### How does the serving function work\n", + "\n", + "When you send a request to an online prediction server, the request is received by a HTTP server. The HTTP server extracts the prediction request from the HTTP request content body. The extracted prediction request is forwarded to the serving function. For Google pre-built prediction containers, the request content is passed to the serving function as a `tf.string`.\n", + "\n", + "The serving function consists of two parts:\n", + "\n", + "- `preprocessing function`:\n", + " - Converts the input (`tf.string`) to the input shape and data type of the underlying model (dynamic graph).\n", + " - Performs the same preprocessing of the data that was done during training the underlying model -- e.g., normalizing, scaling, etc.\n", + "- `post-processing function`:\n", + " - Converts the model output to format expected by the receiving application -- e.q., compresses the output.\n", + " - Packages the output for the the receiving application -- e.g., add headings, make JSON object, etc.\n", + "\n", + "Both the preprocessing and post-processing functions are converted to static graphs which are fused to the model. The output from the underlying model is passed to the post-processing function. The post-processing function passes the converted/packaged output back to the HTTP server. The HTTP server returns the output as the HTTP response content.\n", + "\n", + "One consideration you need to consider when building serving functions for TF.Keras models is that they run as static graphs. That means, you cannot use TF graph operations that require a dynamic graph. If you do, you will get an error during the compile of the serving function which will indicate that you are using an EagerTensor which is not supported." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_image" + }, + "source": [ + "### Serving function for image data\n", + "\n", + "To pass images to the prediction service, you encode the compressed (e.g., JPEG) image bytes into base 64 -- which makes the content safe from modification while transmitting binary data over the network. Since this deployed model expects input data as raw (uncompressed) bytes, you need to ensure that the base 64 encoded data gets converted back to raw bytes before it is passed as input to the deployed model.\n", + "\n", + "To resolve this, define a serving function (`serving_fn`) and attach it to the model as a preprocessing step. Add a `@tf.function` decorator so the serving function is fused to the underlying model (instead of upstream on a CPU).\n", + "\n", + "When you send a prediction or explanation request, the content of the request is base 64 decoded into a Tensorflow string (`tf.string`), which is passed to the serving function (`serving_fn`). The serving function preprocesses the `tf.string` into raw (uncompressed) numpy bytes (`preprocess_fn`) to match the input requirements of the model:\n", + "- `io.decode_jpeg`- Decompresses the JPG image which is returned as a Tensorflow tensor with three channels (RGB).\n", + "- `image.convert_image_dtype` - Changes integer pixel values to float 32.\n", + "- `image.resize` - Resizes the image to match the input shape for the model.\n", + "- `resized / 255.0` - Rescales (normalization) the pixel data between 0 and 1.\n", + "\n", + "At this point, the data can be passed to the model (`m_call`)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_image" + }, + "outputs": [], + "source": [ + "CONCRETE_INPUT = \"numpy_inputs\"\n", + "\n", + "\n", + "def _preprocess(bytes_input):\n", + " decoded = tf.io.decode_jpeg(bytes_input, channels=3)\n", + " decoded = tf.image.convert_image_dtype(decoded, tf.float32)\n", + " resized = tf.image.resize(decoded, size=(32, 32))\n", + " rescale = tf.cast(resized / 255.0, tf.float32)\n", + " return rescale\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def preprocess_fn(bytes_inputs):\n", + " decoded_images = tf.map_fn(\n", + " _preprocess, bytes_inputs, dtype=tf.float32, back_prop=False\n", + " )\n", + " return {\n", + " CONCRETE_INPUT: decoded_images\n", + " } # User needs to make sure the key matches model's input\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def serving_fn(bytes_inputs):\n", + " images = preprocess_fn(bytes_inputs)\n", + " prob = m_call(**images)\n", + " return prob\n", + "\n", + "\n", + "m_call = tf.function(model.call).get_concrete_function(\n", + " [tf.TensorSpec(shape=[None, 32, 32, 3], dtype=tf.float32, name=CONCRETE_INPUT)]\n", + ")\n", + "\n", + "tf.saved_model.save(\n", + " model, model_path_to_deploy, signatures={\"serving_default\": serving_fn}\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_signature:image" + }, + "source": [ + "## Get the serving function signature\n", + "\n", + "You can get the signatures of your model's input and output layers by reloading the model into memory, and querying it for the signatures corresponding to each layer.\n", + "\n", + "For your purpose, you need the signature of the serving function. Why? Well, when we send our data for prediction as a HTTP request packet, the image data is base64 encoded, and our TF.Keras model takes numpy input. Your serving function will do the conversion from base64 to a numpy array.\n", + "\n", + "When making a prediction request, you need to route the request to the serving function instead of the model, so you need to know the input layer name of the serving function -- which you will use later when you make a prediction request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_signature:image" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_path_to_deploy)\n", + "\n", + "serving_input = list(\n", + " loaded.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "print(\"Serving function input:\", serving_input)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "upload_the_model:tfserving" + }, + "source": [ + "### Upload the model\n", + "\n", + "Use this helper function `upload_model` to upload your model, stored in SavedModel format, up to the `Model` service, which will instantiate a Vertex `Model` resource instance for your model. Once you've done that, you can use the `Model` resource instance in the same way as any other Vertex `Model` resource instance, such as deploying to an `Endpoint` resource for serving predictions.\n", + "\n", + "The helper function takes the following parameters:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` service.\n", + "- `image_uri`: The container image for the model deployment.\n", + "- `model_uri`: The Cloud Storage path to our SavedModel artificat. For this tutorial, this is the Cloud Storage location where the `trainer/task.py` saved the model artifacts, which we specified in the variable `MODEL_DIR`.\n", + "\n", + "The helper function calls the `Model` client service's method `upload_model`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for `Dataset`, `Model` and `Endpoint` resources.\n", + "- `model`: The specification for the Vertex `Model` resource instance.\n", + "\n", + "Let's now dive deeper into the Vertex model specification `model`. This is a dictionary object that consists of the following fields:\n", + "\n", + "- `display_name`: A human readable name for the `Model` resource.\n", + "- `metadata_schema_uri`: Since your model was built without an Vertex `Dataset` resource, you will leave this blank (`''`).\n", + "- `artificat_uri`: The Cloud Storage path where the model is stored in SavedModel format.\n", + "- `container_spec`: This is the specification for the Docker container that will be installed on the endpoint, from which the model will serve predictions. In this tutoriak, you use a TF Serving as a custom prediction container. Since this is a custom container, you need to tell Vertex how to start up and communicate with the serving binary:\n", + " - `command`: The serving binary (HTTP Server) to start up.\n", + " - `args`: The arguments to pass to the serving binary. For TF Serving, the required arguments are:\n", + " - `--model_name`: The human readable name to assign to the model.\n", + " - `--model_base_name`: Where to store the model artifacts in the container. The Vertex service sets the variable $(AIP_STORAGE_URI) to where the service installed the model artifacts in the container.\n", + " - `--rest_api_port`: The port to which to send REST based prediction requests. Can either be 8080 or 8501 (default for TF Serving).\n", + " - `--port`: The port to which to send gRPC based prediction requests. Should be 8500 for TF Serving.\n", + "\n", + " - `health_route`: The URL for the service to periodically ping for a response to verify that the serving binary is running. For TF Serving, this will be /v1/models/[model_name].\n", + " - `predict_route`: The URL for the service to route REST-based prediction requests to. For TF Serving, this will be /v1/models/[model_name]:predict.\n", + "\n", + "Uploading a model into a Vertex Model resource returns a long running operation, since it may take a few moments. You call response.result(), which is a synchronous call and will return when the Vertex Model resource is ready.\n", + "\n", + "The helper function returns the Vertex fully qualified identifier for the corresponding Vertex Model instance upload_model_response.model. You will save the identifier for subsequent steps in the variable model_to_deploy_id.\n", + "\n", + "*Note*: You drop the ending number subfolder (e.g., /1) from the model path to upload. The Vertex service will upload the parent folder above the subfolder with the model artifacts -- which is what TensorFlow serving binary expects." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "upload_the_model:tfserving" + }, + "outputs": [], + "source": [ + "IMAGE_URI = DEPLOY_IMAGE\n", + "MODEL_NAME = \"cifar10-\" + TIMESTAMP\n", + "\n", + "\n", + "def upload_model(display_name, image_uri, model_uri):\n", + " model = {\n", + " \"display_name\": display_name,\n", + " \"metadata_schema_uri\": \"\",\n", + " \"artifact_uri\": model_uri,\n", + " \"container_spec\": {\n", + " \"image_uri\": image_uri,\n", + " \"command\": [\"/usr/bin/tensorflow_model_server\"],\n", + " \"args\": [\n", + " \"--model_name=\" + MODEL_NAME,\n", + " \"--model_base_path=\" + \"$(AIP_STORAGE_URI)\",\n", + " \"--rest_api_port=8080\",\n", + " \"--port=8500\",\n", + " \"--file_system_poll_wait_seconds=31540000\",\n", + " ],\n", + " \"health_route\": \"/v1/models/\" + MODEL_NAME,\n", + " \"predict_route\": \"/v1/models/\" + MODEL_NAME + \":predict\",\n", + " },\n", + " }\n", + " response = clients[\"model\"].upload_model(parent=PARENT, model=model)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " upload_model_response = response.result(timeout=180)\n", + " print(\"upload_model_response\")\n", + " print(\" model:\", upload_model_response.model)\n", + " return upload_model_response.model\n", + "\n", + "\n", + "model_to_deploy_id = upload_model(\n", + " \"cifar10-\" + TIMESTAMP, IMAGE_URI, model_path_to_deploy[:-2]\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_model" + }, + "source": [ + "### Get `Model` resource information\n", + "\n", + "Now let's get the model information for just your model. Use this helper function `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource.\n", + "\n", + "This helper function calls the Vertex `Model` client service's method `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_model" + }, + "outputs": [], + "source": [ + "def get_model(name):\n", + " response = clients[\"model\"].get_model(name=name)\n", + " print(response)\n", + "\n", + "\n", + "get_model(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:custom" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex custom `Model` resource. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"cifar10_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:dedicated" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `dedicated_resources`: This refers to how many compute instances (replicas) that are scaled for serving prediction requests.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `min_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:dedicated" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"cifar10_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " \"machine_spec\": machine_spec,\n", + " },\n", + " \"disable_container_logging\": False,\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_item:test" + }, + "source": [ + "### Get test item\n", + "\n", + "You will use an example out of the test (holdout) portion of the dataset as a test item." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:test" + }, + "outputs": [], + "source": [ + "test_image = x_test[0]\n", + "test_label = y_test[0]\n", + "print(test_image.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "prepare_test_item:test,image" + }, + "source": [ + "### Prepare the request content\n", + "You are going to send the CIFAR10 image as compressed JPG image, instead of the raw uncompressed bytes:\n", + "\n", + "- `cv2.imwrite`: Use openCV to write the uncompressed image to disk as a compressed JPEG image.\n", + " - Denormalize the image data from \\[0,1) range back to [0,255).\n", + " - Convert the 32-bit floating point values to 8-bit unsigned integers.\n", + "- `tf.io.read_file`: Read the compressed JPG images back into memory as raw bytes.\n", + "- `base64.b64encode`: Encode the raw bytes into a base 64 encoded string." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "prepare_test_item:test,image" + }, + "outputs": [], + "source": [ + "import base64\n", + "\n", + "import cv2\n", + "\n", + "cv2.imwrite(\"tmp.jpg\", (test_image * 255).astype(np.uint8))\n", + "\n", + "bytes = tf.io.read_file(\"tmp.jpg\")\n", + "b64str = base64.b64encode(bytes.numpy()).decode(\"utf-8\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "send_prediction_request:image" + }, + "source": [ + "### Send the prediction request\n", + "\n", + "Ok, now you have a test image. Use this helper function `predict_image`, which takes the following parameters:\n", + "\n", + "- `image`: The test image data as a numpy array.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed to.\n", + "- `parameters_dict`: Additional parameters for serving.\n", + "\n", + "This function calls the prediction client service `predict` method with the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed to.\n", + "- `instances`: A list of instances (encoded images) to predict.\n", + "- `parameters`: Additional parameters for serving.\n", + "\n", + "To pass the image data to the prediction service, in the previous step you encoded the bytes into base64 -- which makes the content safe from modification when transmitting binary data over the network. You need to tell the serving binary where your model is deployed to, that the content has been base64 encoded, so it will decode it on the other end in the serving binary.\n", + "\n", + "Each instance in the prediction request is a dictionary entry of the form:\n", + "\n", + " {serving_input: {'b64': content}}\n", + "\n", + "- `input_name`: the name of the input layer of the underlying model.\n", + "- `'b64'`: A key that indicates the content is base64 encoded.\n", + "- `content`: The compressed JPG image bytes as a base64 encoded string.\n", + "\n", + "Since the `predict()` service can take multiple images (instances), you will send your single image as a list of one image. As a final step, you package the instances list into Google's protobuf format -- which is what we pass to the `predict()` service.\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding image in the request. You will see in the output for each prediction:\n", + "\n", + "- `predictions`: Confidence level for the prediction, between 0 and 1, for each of the classes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "send_prediction_request:image" + }, + "outputs": [], + "source": [ + "def predict_image(image, endpoint, parameters_dict):\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [{serving_input: {\"b64\": image}}]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters_dict\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", prediction)\n", + "\n", + "\n", + "predict_image(b64str, endpoint_id, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_custom_image_classification_online_tfserving.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_custom_image_super_resolution_online_post.ipynb b/notebooks/community/gapic/custom/showcase_custom_image_super_resolution_online_post.ipynb new file mode 100644 index 000000000..6ff6fe4b9 --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_custom_image_super_resolution_online_post.ipynb @@ -0,0 +1,2188 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Custom training image super resolution model for online prediction with post processing of prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom,post" + }, + "source": [ + "## Overview\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to train and deploy a custom image super resolution model for online prediction, with post-processing of the prediction." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:custom,cifar10,icn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [CIFAR10 dataset](https://www.tensorflow.org/datasets/catalog/cifar10) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use is built into TensorFlow. The trained model predicts which type of class an image is from ten classes: airplane, automobile, bird, cat, deer, dog, frog, horse, ship, truck." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,post,online_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you will learn how to create a custom model from a Python script in a Docker container using the Vertex client library, do a prediction on the deployed model, and do post-processing on the prediction in the serving binary. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex custom job for training a model.\n", + "- Train a TensorFlow model using a custom container.\n", + "- Retrieve and load the model artifacts.\n", + "- View the model evaluation.\n", + "- Construct serving function for post processing.\n", + "- Upload the model as a Vertex `Model` resource.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for training and prediction.\n", + "\n", + "Set the variables `TRAIN_GPU/TRAIN_NGPU` and `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU.\n", + "\n", + "*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3 -- which is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n", + " TRAIN_GPU, TRAIN_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n", + " )\n", + "else:\n", + " TRAIN_GPU, TRAIN_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (None, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:training,prediction" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for training and prediction\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-3:latest`\n", + " - TensorFlow 2.4\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-4:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/training/xgboost-cpu.1-1`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/training/scikit-learn-cpu.0-23:latest`\n", + " - Pytorch\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-5:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-6:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-7:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for training](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers).\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-3:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-90:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-82:latest`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-23:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-22:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-20:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for prediction](https://cloud.google.com/vertex-ai/docs/predictions/pre-built-containers)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf2-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf2-cpu.{}\".format(TF)\n", + "else:\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "TRAIN_IMAGE = \"gcr.io/cloud-aiplatform/training/{}:latest\".format(TRAIN_VERSION)\n", + "DEPLOY_IMAGE = \"gcr.io/cloud-aiplatform/prediction/{}:latest\".format(DEPLOY_VERSION)\n", + "\n", + "print(\"Training:\", TRAIN_IMAGE, TRAIN_GPU, TRAIN_NGPU)\n", + "print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU, DEPLOY_NGPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:training,prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for training and prediction.\n", + "\n", + "- Set the variables `TRAIN_COMPUTE` and `DEPLOY_COMPUTE` to configure the compute resources for the VMs you will use for for training and prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: The following is not supported for training:*\n", + "\n", + " - `standard`: 2 vCPUs\n", + " - `highcpu`: 2, 4 and 8 vCPUs\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_TRAIN_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Train machine type\", TRAIN_COMPUTE)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own custom model and training for CIFAR10." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:custom" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Model Service for `Model` resources.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:custom" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"job\"] = create_job_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_model" + }, + "source": [ + "## Train a model\n", + "\n", + "There are two ways you can train a custom model using a container image:\n", + "\n", + "- **Use a Google Cloud prebuilt container**. If you use a prebuilt container, you will additionally specify a Python package to install into the container image. This Python package contains your code for training a custom model.\n", + "\n", + "- **Use your own custom container image**. If you use your own container, the container needs to contain your code for training a custom model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_specification:prebuilt_container" + }, + "source": [ + "## Prepare your custom job specification\n", + "\n", + "Now that your clients are ready, your first step is to create a Job Specification for your custom training job. The job specification will consist of the following:\n", + "\n", + "- `worker_pool_spec` : The specification of the type of machine(s) you will use for training and how many (single or distributed)\n", + "- `python_package_spec` : The specification of the Python package to be installed with the pre-built container." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "source": [ + "### Prepare your machine specification\n", + "\n", + "Now define the machine specification for your custom training job. This tells Vertex what type of machine instance to provision for the training.\n", + " - `machine_type`: The type of GCP instance to provision -- e.g., n1-standard-8.\n", + " - `accelerator_type`: The type, if any, of hardware accelerator. In this tutorial if you previously set the variable `TRAIN_GPU != None`, you are using a GPU; otherwise you will use a CPU.\n", + " - `accelerator_count`: The number of accelerators." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "outputs": [], + "source": [ + "if TRAIN_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": TRAIN_COMPUTE,\n", + " \"accelerator_type\": TRAIN_GPU,\n", + " \"accelerator_count\": TRAIN_NGPU,\n", + " }\n", + "else:\n", + " machine_spec = {\"machine_type\": TRAIN_COMPUTE, \"accelerator_count\": 0}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "source": [ + "### Prepare your disk specification\n", + "\n", + "(optional) Now define the disk specification for your custom training job. This tells Vertex what type and size of disk to provision in each machine instance for the training.\n", + "\n", + " - `boot_disk_type`: Either SSD or Standard. SSD is faster, and Standard is less expensive. Defaults to SSD.\n", + " - `boot_disk_size_gb`: Size of disk in GB." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "outputs": [], + "source": [ + "DISK_TYPE = \"pd-ssd\" # [ pd-ssd, pd-standard]\n", + "DISK_SIZE = 200 # GB\n", + "\n", + "disk_spec = {\"boot_disk_type\": DISK_TYPE, \"boot_disk_size_gb\": DISK_SIZE}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container" + }, + "source": [ + "### Define the worker pool specification\n", + "\n", + "Next, you define the worker pool specification for your custom training job. The worker pool specification will consist of the following:\n", + "\n", + "- `replica_count`: The number of instances to provision of this machine type.\n", + "- `machine_spec`: The hardware specification.\n", + "- `disk_spec` : (optional) The disk storage specification.\n", + "\n", + "- `python_package`: The Python training package to install on the VM instance(s) and which Python module to invoke, along with command line arguments for the Python module.\n", + "\n", + "Let's dive deeper now into the python package specification:\n", + "\n", + "-`executor_image_spec`: This is the docker image which is configured for your custom training job.\n", + "\n", + "-`package_uris`: This is a list of the locations (URIs) of your python training packages to install on the provisioned instance. The locations need to be in a Cloud Storage bucket. These can be either individual python files or a zip (archive) of an entire package. In the later case, the job service will unzip (unarchive) the contents into the docker image.\n", + "\n", + "-`python_module`: The Python module (script) to invoke for running the custom training job. In this example, you will be invoking `trainer.task.py` -- note that it was not neccessary to append the `.py` suffix.\n", + "\n", + "-`args`: The command line arguments to pass to the corresponding Pythom module. In this example, you will be setting:\n", + " - `\"--model-dir=\" + MODEL_DIR` : The Cloud Storage location where to store the model artifacts. There are two ways to tell the training script where to save the model artifacts:\n", + " - direct: You pass the Cloud Storage location as a command line argument to your training script (set variable `DIRECT = True`), or\n", + " - indirect: The service passes the Cloud Storage location as the environment variable `AIP_MODEL_DIR` to your training script (set variable `DIRECT = False`). In this case, you tell the service the model artifact location in the job specification.\n", + " - `\"--epochs=\" + EPOCHS`: The number of epochs for training.\n", + " - `\"--steps=\" + STEPS`: The number of steps (batches) per epoch.\n", + " - `\"--distribute=\" + TRAIN_STRATEGY\"` : The training distribution strategy to use for single or distributed training.\n", + " - `\"single\"`: single device.\n", + " - `\"mirror\"`: all GPU devices on a single compute instance.\n", + " - `\"multi\"`: all GPU devices on all compute instances." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_job_\" + TIMESTAMP\n", + "MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, JOB_NAME)\n", + "\n", + "if not TRAIN_NGPU or TRAIN_NGPU < 2:\n", + " TRAIN_STRATEGY = \"single\"\n", + "else:\n", + " TRAIN_STRATEGY = \"mirror\"\n", + "\n", + "EPOCHS = 20\n", + "STEPS = 100\n", + "\n", + "DIRECT = True\n", + "if DIRECT:\n", + " CMDARGS = [\n", + " \"--model-dir=\" + MODEL_DIR,\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "else:\n", + " CMDARGS = [\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "\n", + "worker_pool_spec = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": machine_spec,\n", + " \"disk_spec\": disk_spec,\n", + " \"python_package_spec\": {\n", + " \"executor_image_uri\": TRAIN_IMAGE,\n", + " \"package_uris\": [BUCKET_NAME + \"/trainer_cifar10.tar.gz\"],\n", + " \"python_module\": \"trainer.task\",\n", + " \"args\": CMDARGS,\n", + " },\n", + " }\n", + "]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "assemble_custom_job_specification" + }, + "source": [ + "### Assemble a job specification\n", + "\n", + "Now assemble the complete description for the custom job specification:\n", + "\n", + "- `display_name`: The human readable name you assign to this custom job.\n", + "- `job_spec`: The specification for the custom job.\n", + " - `worker_pool_specs`: The specification for the machine VM instances.\n", + " - `base_output_directory`: This tells the service the Cloud Storage location where to save the model artifacts (when variable `DIRECT = False`). The service will then pass the location to the training script as the environment variable `AIP_MODEL_DIR`, and the path will be of the form:\n", + "\n", + " /model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "assemble_custom_job_specification" + }, + "outputs": [], + "source": [ + "if DIRECT:\n", + " job_spec = {\"worker_pool_specs\": worker_pool_spec}\n", + "else:\n", + " job_spec = {\n", + " \"worker_pool_specs\": worker_pool_spec,\n", + " \"base_output_directory\": {\"output_uri_prefix\": MODEL_DIR},\n", + " }\n", + "\n", + "custom_job = {\"display_name\": JOB_NAME, \"job_spec\": job_spec}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "examine_training_package" + }, + "source": [ + "### Examine the training package\n", + "\n", + "#### Package layout\n", + "\n", + "Before you start the training, you will look at how a Python package is assembled for a custom training job. When unarchived, the package contains the following directory/file layout.\n", + "\n", + "- PKG-INFO\n", + "- README.md\n", + "- setup.cfg\n", + "- setup.py\n", + "- trainer\n", + " - \\_\\_init\\_\\_.py\n", + " - task.py\n", + "\n", + "The files `setup.cfg` and `setup.py` are the instructions for installing the package into the operating environment of the Docker image.\n", + "\n", + "The file `trainer/task.py` is the Python script for executing the custom training job. *Note*, when we referred to it in the worker pool specification, we replace the directory slash with a dot (`trainer.task`) and dropped the file suffix (`.py`).\n", + "\n", + "#### Package Assembly\n", + "\n", + "In the following cells, you will assemble the training package." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "examine_training_package" + }, + "outputs": [], + "source": [ + "# Make folder for Python training script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\ntag_build =\\n\\ntag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'tensorflow_datasets==1.3.0',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\nName: CIFAR10 image super resolution\\n\\nVersion: 0.0.0\\n\\nSummary: Demostration training script\\n\\nHome-page: www.google.com\\n\\nAuthor: Google\\n\\nAuthor-email: aferlitsch@google.com\\n\\nLicense: Public\\n\\nDescription: Demo\\n\\nPlatform: Vertex\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:cifar10" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the training script task.py. We won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n", + "- Loads CIFAR10 dataset from TF Datasets (tfds).\n", + "- Builds a model using TF.Keras model API.\n", + "- Compiles the model (`compile()`).\n", + "- Sets a training distribution strategy according to the argument `args.distribute`.\n", + "- Trains the model (`fit()`) with epochs and steps according to the arguments `args.epochs` and `args.steps`\n", + "- Saves the trained model (`save(args.model_dir)`) to the specified model directory." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:cifar10" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Single, Mirror and Multi-Machine Distributed Training for CIFAR-10\n", + "\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "import argparse\n", + "import os\n", + "import sys\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv(\"AIP_MODEL_DIR\"), type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.01, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=10, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=200, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "print('DEVICES', device_lib.list_local_devices())\n", + "\n", + "# Single Machine, single compute device\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "# Single Machine, multiple compute device\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "# Multiple Machine, multiple compute device\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "# Multi-worker configuration\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "# Preparing dataset\n", + "BUFFER_SIZE = 10000\n", + "BATCH_SIZE = 64\n", + "\n", + "\n", + "def make_datasets_unbatched():\n", + "\n", + " # Scaling CIFAR10 data from (0, 255] to (0., 1.]\n", + " def scale(image, label):\n", + " image = tf.cast(image, tf.float32)\n", + " image /= 255.0\n", + " return image, label\n", + "\n", + "\n", + " datasets, info = tfds.load(name='cifar10',\n", + " with_info=True,\n", + " as_supervised=True)\n", + " return datasets['train'].map(scale).cache().shuffle(BUFFER_SIZE).repeat()\n", + "\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_cnn_model():\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu', input_shape=(32, 32, 3)),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu'),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Flatten(),\n", + " tf.keras.layers.Dense(10, activation='softmax')\n", + " ])\n", + " model.compile(\n", + " loss=tf.keras.losses.sparse_categorical_crossentropy,\n", + " optimizer=tf.keras.optimizers.SGD(learning_rate=args.lr),\n", + " metrics=['accuracy'])\n", + " return model\n", + "\n", + "\n", + "# Train the model\n", + "NUM_WORKERS = strategy.num_replicas_in_sync\n", + "# Here the batch size scales up by number of workers since\n", + "# `tf.data.Dataset.batch` expects the global batch size.\n", + "GLOBAL_BATCH_SIZE = BATCH_SIZE * NUM_WORKERS\n", + "train_dataset = make_datasets_unbatched().batch(GLOBAL_BATCH_SIZE)\n", + "\n", + "with strategy.scope():\n", + " # Creation of dataset, and model building/compiling need to be within\n", + " # `strategy.scope()`.\n", + " model = build_and_compile_cnn_model()\n", + "\n", + "model.fit(x=train_dataset, epochs=args.epochs, steps_per_epoch=args.steps)\n", + "model.save(args.model_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tarball_training_script" + }, + "source": [ + "#### Store training script on your Cloud Storage bucket\n", + "\n", + "Next, you package the training folder into a compressed tar ball, and then store it in your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tarball_training_script" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_cifar10.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job" + }, + "source": [ + "### Train the model\n", + "\n", + "\n", + "Now start the training of your custom training job on Vertex. Use this helper function `create_custom_job`, which takes the following parameter:\n", + "\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "The helper function calls job client service's `create_custom_job` method, with the following parameters:\n", + "\n", + "-`parent`: The Vertex location path to `Dataset`, `Model` and `Endpoint` resources.\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "You will display a handful of the fields returned in `response` object, with the two that are of most interest are:\n", + "\n", + "`response.name`: The Vertex fully qualified identifier assigned to this custom training job. You save this identifier for using in subsequent steps.\n", + "\n", + "`response.state`: The current state of the custom training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job" + }, + "outputs": [], + "source": [ + "def create_custom_job(custom_job):\n", + " response = clients[\"job\"].create_custom_job(parent=PARENT, custom_job=custom_job)\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = create_custom_job(custom_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "job_id:response" + }, + "source": [ + "Now get the unique identifier for the custom job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom job\n", + "job_id = response.name\n", + "# The short numeric ID for the custom job\n", + "job_short_id = job_id.split(\"/\")[-1]\n", + "\n", + "print(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_custom_job" + }, + "source": [ + "### Get information on a custom job\n", + "\n", + "Next, use this helper function `get_custom_job`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "The helper function calls the job client service's`get_custom_job` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "If you recall, you got the Vertex fully qualified identifier for the custom job in the `response.name` field when you called the `create_custom_job` method, and saved the identifier in the variable `job_id`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_custom_job" + }, + "outputs": [], + "source": [ + "def get_custom_job(name, silent=False):\n", + " response = clients[\"job\"].get_custom_job(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = get_custom_job(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete:custom" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, we will need to know the location of the saved model, which the Python script saved in your local Cloud Storage bucket at `MODEL_DIR + '/saved_model.pb'`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete:custom" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_custom_job(job_id, True)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_path_to_deploy = None\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " if not DIRECT:\n", + " MODEL_DIR = MODEL_DIR + \"/model\"\n", + " model_path_to_deploy = MODEL_DIR\n", + " print(\"Training Time:\", response.update_time - response.create_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model_to_deploy:\", model_path_to_deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "load_saved_model" + }, + "source": [ + "## Load the saved model\n", + "\n", + "Your model is stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Now load it from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n", + "\n", + "To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "load_saved_model" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:image" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load the CIFAR10 test (holdout) data from `tf.keras.datasets`, using the method `load_data()`. This will return the dataset as a tuple of two elements. The first element is the training data and the second is the test data. Each element is also a tuple of two elements: the image data, and the corresponding labels.\n", + "\n", + "You don't need the training data, and hence why we loaded it as `(_, _)`.\n", + "\n", + "Before you can run the data through evaluation, you need to preprocess it:\n", + "\n", + "x_test:\n", + "1. Normalize (rescaling) the pixel data by dividing each pixel by 255. This will replace each single byte integer pixel with a 32-bit floating point number between 0 and 1.\n", + "\n", + "y_test:
\n", + "2. The labels are currently scalar (sparse). If you look back at the `compile()` step in the `trainer/task.py` script, you will find that it was compiled for sparse labels. So we don't need to do anything more." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "evaluate_custom_model:image,cifar10" + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "from tensorflow.keras.datasets import cifar10\n", + "\n", + "(_, _), (x_test, y_test) = cifar10.load_data()\n", + "x_test = (x_test / 255.0).astype(np.float32)\n", + "\n", + "print(x_test.shape, y_test.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "perform_evaluation_custom" + }, + "source": [ + "### Perform the model evaluation\n", + "\n", + "Now evaluate how well the model in the custom job did." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "perform_evaluation_custom" + }, + "outputs": [], + "source": [ + "model.evaluate(x_test, y_test)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "how_serving_function_works" + }, + "source": [ + "## Upload the model for serving\n", + "\n", + "Next, you will upload your TF.Keras model from the custom job to Vertex `Model` service, which will create a Vertex `Model` resource for your custom model. During upload, you need to define a serving function to convert data to the format your model expects. If you send encoded data to Vertex, your serving function ensures that the data is decoded on the model server before it is passed as input to your model.\n", + "\n", + "### How does the serving function work\n", + "\n", + "When you send a request to an online prediction server, the request is received by a HTTP server. The HTTP server extracts the prediction request from the HTTP request content body. The extracted prediction request is forwarded to the serving function. For Google pre-built prediction containers, the request content is passed to the serving function as a `tf.string`.\n", + "\n", + "The serving function consists of two parts:\n", + "\n", + "- `preprocessing function`:\n", + " - Converts the input (`tf.string`) to the input shape and data type of the underlying model (dynamic graph).\n", + " - Performs the same preprocessing of the data that was done during training the underlying model -- e.g., normalizing, scaling, etc.\n", + "- `post-processing function`:\n", + " - Converts the model output to format expected by the receiving application -- e.q., compresses the output.\n", + " - Packages the output for the the receiving application -- e.g., add headings, make JSON object, etc.\n", + "\n", + "Both the preprocessing and post-processing functions are converted to static graphs which are fused to the model. The output from the underlying model is passed to the post-processing function. The post-processing function passes the converted/packaged output back to the HTTP server. The HTTP server returns the output as the HTTP response content.\n", + "\n", + "One consideration you need to consider when building serving functions for TF.Keras models is that they run as static graphs. That means, you cannot use TF graph operations that require a dynamic graph. If you do, you will get an error during the compile of the serving function which will indicate that you are using an EagerTensor which is not supported." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_image:post" + }, + "source": [ + "### Serving function for image data\n", + "\n", + "#### Preprocessing\n", + "\n", + "To pass images to the prediction service, you encode the compressed (e.g., JPEG) image bytes into base 64 -- which makes the content safe from modification while transmitting binary data over the network. Since this deployed model expects input data as raw (uncompressed) bytes, you need to ensure that the base 64 encoded data gets converted back to raw bytes before it is passed as input to the deployed model.\n", + "\n", + "To resolve this, define a serving function (`serving_fn`) and attach it to the model as a preprocessing step. Add a `@tf.function` decorator so the serving function is fused to the underlying model (instead of upstream on a CPU).\n", + "\n", + "When you send a prediction or explanation request, the content of the request is base 64 decoded into a Tensorflow string (`tf.string`), which is passed to the serving function (`serving_fn`). The serving function preprocesses the `tf.string` into raw (uncompressed) numpy bytes (`preprocess_fn`) to match the input requirements of the model:\n", + "- `io.decode_jpeg`- Decompresses the JPG image which is returned as a Tensorflow tensor with three channels (RGB).\n", + "- `image.convert_image_dtype` - Changes integer pixel values to float 32.\n", + "- `image.resize` - Resizes the image to match the input shape for the model.\n", + "- `resized / 255.0` - Rescales (normalization) the pixel data between 0 and 1.\n", + "\n", + "At this point, the data can be passed to the model (`m_call`).\n", + "\n", + "#### Post-Processing\n", + "\n", + "The return value from `prob = m_call(**images)` will be a list of tensors, one per instance in the prediction request. Each tensor will be the predicted super-resolution image of the corresponding instance.\n", + "\n", + "**TODO**" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_image:post" + }, + "outputs": [], + "source": [ + "CONCRETE_INPUT = \"numpy_inputs\"\n", + "\n", + "\n", + "def _preprocess(bytes_input):\n", + " decoded = tf.io.decode_jpeg(bytes_input, channels=3)\n", + " decoded = tf.image.convert_image_dtype(decoded, tf.float32)\n", + " resized = tf.image.resize(decoded, size=(16, 16))\n", + " rescale = tf.cast(resized / 255.0, tf.float32)\n", + " return rescale\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def preprocess_fn(bytes_inputs):\n", + " decoded_images = tf.map_fn(\n", + " _preprocess, bytes_inputs, dtype=tf.float32, back_prop=False\n", + " )\n", + " return {\n", + " CONCRETE_INPUT: decoded_images\n", + " } # User needs to make sure the key matches model's input\n", + "\n", + "\n", + "m_call = tf.function(model.call).get_concrete_function(\n", + " [tf.TensorSpec(shape=[None, 16, 16, 3], dtype=tf.float32, name=CONCRETE_INPUT)]\n", + ")\n", + "\n", + "\n", + "def _postprocess(bytes_output):\n", + " rescale = tf.cast(bytes_output * 255.0, tf.uint8)\n", + " reshape = tf.reshape(rescale, (32, 32, 3))\n", + " encoded = tf.io.encode_jpeg(reshape)\n", + " return tf.cast(encoded, tf.string)\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None, 32, 32, 3], tf.float32)])\n", + "def postprocess_fn(bytes_outputs):\n", + " encoded_images = tf.map_fn(\n", + " _postprocess, bytes_outputs, dtype=tf.string, back_prop=False\n", + " )\n", + " return encoded_images\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def serving_fn(bytes_inputs):\n", + " images = preprocess_fn(bytes_inputs)\n", + " prob = m_call(**images)\n", + " post = postprocess_fn(prob)\n", + " return post\n", + "\n", + "\n", + "tf.saved_model.save(\n", + " model,\n", + " model_path_to_deploy,\n", + " signatures={\n", + " \"serving_default\": serving_fn,\n", + " },\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_signature:image" + }, + "source": [ + "## Get the serving function signature\n", + "\n", + "You can get the signatures of your model's input and output layers by reloading the model into memory, and querying it for the signatures corresponding to each layer.\n", + "\n", + "For your purpose, you need the signature of the serving function. Why? Well, when we send our data for prediction as a HTTP request packet, the image data is base64 encoded, and our TF.Keras model takes numpy input. Your serving function will do the conversion from base64 to a numpy array.\n", + "\n", + "When making a prediction request, you need to route the request to the serving function instead of the model, so you need to know the input layer name of the serving function -- which you will use later when you make a prediction request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_signature:image" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_path_to_deploy)\n", + "\n", + "serving_input = list(\n", + " loaded.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "print(\"Serving function input:\", serving_input)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "upload_the_model" + }, + "source": [ + "### Upload the model\n", + "\n", + "Use this helper function `upload_model` to upload your model, stored in SavedModel format, up to the `Model` service, which will instantiate a Vertex `Model` resource instance for your model. Once you've done that, you can use the `Model` resource instance in the same way as any other Vertex `Model` resource instance, such as deploying to an `Endpoint` resource for serving predictions.\n", + "\n", + "The helper function takes the following parameters:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` service.\n", + "- `image_uri`: The container image for the model deployment.\n", + "- `model_uri`: The Cloud Storage path to our SavedModel artificat. For this tutorial, this is the Cloud Storage location where the `trainer/task.py` saved the model artifacts, which we specified in the variable `MODEL_DIR`.\n", + "\n", + "The helper function calls the `Model` client service's method `upload_model`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for `Dataset`, `Model` and `Endpoint` resources.\n", + "- `model`: The specification for the Vertex `Model` resource instance.\n", + "\n", + "Let's now dive deeper into the Vertex model specification `model`. This is a dictionary object that consists of the following fields:\n", + "\n", + "- `display_name`: A human readable name for the `Model` resource.\n", + "- `metadata_schema_uri`: Since your model was built without an Vertex `Dataset` resource, you will leave this blank (`''`).\n", + "- `artificat_uri`: The Cloud Storage path where the model is stored in SavedModel format.\n", + "- `container_spec`: This is the specification for the Docker container that will be installed on the `Endpoint` resource, from which the `Model` resource will serve predictions. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + "\n", + "Uploading a model into a Vertex Model resource returns a long running operation, since it may take a few moments. You call response.result(), which is a synchronous call and will return when the Vertex Model resource is ready.\n", + "\n", + "The helper function returns the Vertex fully qualified identifier for the corresponding Vertex Model instance upload_model_response.model. You will save the identifier for subsequent steps in the variable model_to_deploy_id." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "upload_the_model" + }, + "outputs": [], + "source": [ + "IMAGE_URI = DEPLOY_IMAGE\n", + "\n", + "\n", + "def upload_model(display_name, image_uri, model_uri):\n", + " model = {\n", + " \"display_name\": display_name,\n", + " \"metadata_schema_uri\": \"\",\n", + " \"artifact_uri\": model_uri,\n", + " \"container_spec\": {\n", + " \"image_uri\": image_uri,\n", + " \"command\": [],\n", + " \"args\": [],\n", + " \"env\": [{\"name\": \"env_name\", \"value\": \"env_value\"}],\n", + " \"ports\": [{\"container_port\": 8080}],\n", + " \"predict_route\": \"\",\n", + " \"health_route\": \"\",\n", + " },\n", + " }\n", + " response = clients[\"model\"].upload_model(parent=PARENT, model=model)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " upload_model_response = response.result(timeout=180)\n", + " print(\"upload_model_response\")\n", + " print(\" model:\", upload_model_response.model)\n", + " return upload_model_response.model\n", + "\n", + "\n", + "model_to_deploy_id = upload_model(\n", + " \"cifar10-\" + TIMESTAMP, IMAGE_URI, model_path_to_deploy\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_model" + }, + "source": [ + "### Get `Model` resource information\n", + "\n", + "Now let's get the model information for just your model. Use this helper function `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource.\n", + "\n", + "This helper function calls the Vertex `Model` client service's method `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_model" + }, + "outputs": [], + "source": [ + "def get_model(name):\n", + " response = clients[\"model\"].get_model(name=name)\n", + " print(response)\n", + "\n", + "\n", + "get_model(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:custom" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex custom `Model` resource. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"cifar10_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:dedicated" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `dedicated_resources`: This refers to how many compute instances (replicas) that are scaled for serving prediction requests.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `min_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:dedicated" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"cifar10_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " \"machine_spec\": machine_spec,\n", + " },\n", + " \"disable_container_logging\": False,\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_item:test,srn" + }, + "source": [ + "### Get test item\n", + "\n", + "You will use an example out of the test (holdout) portion of the dataset as a test item." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:test,srn" + }, + "outputs": [], + "source": [ + "test_image = x_test_lr[0]\n", + "test_label = y_test[0]\n", + "print(test_image.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "send_prediction_request:image" + }, + "source": [ + "### Send the prediction request\n", + "\n", + "Ok, now you have a test image. Use this helper function `predict_image`, which takes the following parameters:\n", + "\n", + "- `image`: The test image data as a numpy array.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed to.\n", + "- `parameters_dict`: Additional parameters for serving.\n", + "\n", + "This function calls the prediction client service `predict` method with the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed to.\n", + "- `instances`: A list of instances (encoded images) to predict.\n", + "- `parameters`: Additional parameters for serving.\n", + "\n", + "To pass the image data to the prediction service, in the previous step you encoded the bytes into base64 -- which makes the content safe from modification when transmitting binary data over the network. You need to tell the serving binary where your model is deployed to, that the content has been base64 encoded, so it will decode it on the other end in the serving binary.\n", + "\n", + "Each instance in the prediction request is a dictionary entry of the form:\n", + "\n", + " {serving_input: {'b64': content}}\n", + "\n", + "- `input_name`: the name of the input layer of the underlying model.\n", + "- `'b64'`: A key that indicates the content is base64 encoded.\n", + "- `content`: The compressed JPG image bytes as a base64 encoded string.\n", + "\n", + "Since the `predict()` service can take multiple images (instances), you will send your single image as a list of one image. As a final step, you package the instances list into Google's protobuf format -- which is what we pass to the `predict()` service.\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding image in the request. You will see in the output for each prediction:\n", + "\n", + "- `predictions`: Confidence level for the prediction, between 0 and 1, for each of the classes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "send_prediction_request:image" + }, + "outputs": [], + "source": [ + "def predict_image(image, endpoint, parameters_dict):\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [{serving_input: {\"b64\": image}}]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters_dict\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", prediction)\n", + "\n", + "\n", + "predict_image(b64str, endpoint_id, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_custom_image_super_resolution_online_post.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_custom_tabular_classification_online_exported_ds.ipynb b/notebooks/community/gapic/custom/showcase_custom_tabular_classification_online_exported_ds.ipynb new file mode 100644 index 000000000..3de8b5893 --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_custom_tabular_classification_online_exported_ds.ipynb @@ -0,0 +1,2339 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Custom training tabular classification model for online prediction using exported dataset\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom,exported_ds" + }, + "source": [ + "## Overview\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to train and deploy a custom tabular classification model for online prediction, using an exported `Dataset` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:iris,lcn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Iris dataset](https://www.tensorflow.org/datasets/catalog/iris) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). This dataset does not require any feature engineering. The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts the type of Iris flower species from a class of three species: setosa, virginica, or versicolor." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,exported_ds,online_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you learn how to create a custom model using an exported `Dataset` resource from a Python script in a Docker container using the Vertex client library, and then do a prediction on the deployed model. You can alternatively create models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Export the `Dataset` resource's manifest.\n", + "- Create a Vertex custom job for training a model.\n", + "- Import the exported dataset manifest.\n", + "- Train the model.\n", + "- Retrieve and load the model artifacts.\n", + "- View the model evaluation.\n", + "- Upload the model as a Vertex `Model` resource.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "labeling_constants:lcn" + }, + "source": [ + "#### Labeling constants\n", + "\n", + "Set constants unique to `Dataset` resource labeling" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "labeling_constants:lcn" + }, + "outputs": [], + "source": [ + "# Tabular Dataset type\n", + "DATA_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/metadata/tables_1.0.0.yaml\"\n", + "# Tabular Labeling type\n", + "LABEL_SCHEMA = (\n", + " \"gs://google-cloud-aiplatform/schema/dataset/ioformat/table_io_format_1.0.0.yaml\"\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for training and prediction.\n", + "\n", + "Set the variables `TRAIN_GPU/TRAIN_NGPU` and `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU.\n", + "\n", + "*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3 -- which is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n", + " TRAIN_GPU, TRAIN_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n", + " )\n", + "else:\n", + " TRAIN_GPU, TRAIN_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (None, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:training,prediction" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for training and prediction\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-3:latest`\n", + " - TensorFlow 2.4\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-4:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/training/xgboost-cpu.1-1`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/training/scikit-learn-cpu.0-23:latest`\n", + " - Pytorch\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-5:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-6:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-7:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for training](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers).\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-3:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-90:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-82:latest`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-23:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-22:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-20:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for prediction](https://cloud.google.com/vertex-ai/docs/predictions/pre-built-containers)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf2-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf2-cpu.{}\".format(TF)\n", + "else:\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "TRAIN_IMAGE = \"gcr.io/cloud-aiplatform/training/{}:latest\".format(TRAIN_VERSION)\n", + "DEPLOY_IMAGE = \"gcr.io/cloud-aiplatform/prediction/{}:latest\".format(DEPLOY_VERSION)\n", + "\n", + "print(\"Training:\", TRAIN_IMAGE, TRAIN_GPU, TRAIN_NGPU)\n", + "print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU, DEPLOY_NGPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:training,prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for training and prediction.\n", + "\n", + "- Set the variables `TRAIN_COMPUTE` and `DEPLOY_COMPUTE` to configure the compute resources for the VMs you will use for for training and prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: The following is not supported for training:*\n", + "\n", + " - `standard`: 2 vCPUs\n", + " - `highcpu`: 2, 4 and 8 vCPUs\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_TRAIN_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Train machine type\", TRAIN_COMPUTE)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own custom model and training for Iris." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:custom,exported_ds" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:custom,exported_ds" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"job\"] = create_job_client()\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "\n", + "def create_dataset(name, schema, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " dataset = aip.Dataset(\n", + " display_name=name, metadata_schema_uri=schema, labels=labels\n", + " )\n", + "\n", + " operation = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"iris-\" + TIMESTAMP, DATA_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:tabular,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for tabular has a couple of requirements for your tabular data.\n", + "\n", + "- Must be in a CSV file or a BigQuery query." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:lcn,u_dataset,csv" + }, + "source": [ + "#### CSV\n", + "\n", + "For tabular classification, the CSV file has a few requirements:\n", + "\n", + "- The first row must be the heading -- note how this is different from Vision, Video and Language where the requirement is no heading.\n", + "- All but one column are features.\n", + "- One column is the label, which you will specify when you subsequently create the training pipeline." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:iris,csv,lcn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://cloud-samples-data/tables/iris_1000.csv\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:csv" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Iris dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:csv" + }, + "outputs": [], + "source": [ + "if \"IMPORT_FILES\" in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_data" + }, + "source": [ + "### Import data\n", + "\n", + "Now, import the data into your Vertex Dataset resource. Use this helper function `import_data` to import the data. The function does the following:\n", + "\n", + "- Uses the `Dataset` client.\n", + "- Calls the client method `import_data`, with the following parameters:\n", + " - `name`: The human readable name you give to the `Dataset` resource (e.g., iris).\n", + " - `import_configs`: The import configuration.\n", + "\n", + "- `import_configs`: A Python list containing a dictionary, with the key/value entries:\n", + " - `gcs_sources`: A list of URIs to the paths of the one or more index files.\n", + " - `import_schema_uri`: The schema identifying the labeling type.\n", + "\n", + "The `import_data()` method returns a long running `operation` object. This will take a few minutes to complete. If you are in a live tutorial, this would be a good time to ask questions, or take a personal break." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_data" + }, + "outputs": [], + "source": [ + "def import_data(dataset, gcs_sources, schema):\n", + " config = [{\"gcs_source\": {\"uris\": gcs_sources}, \"import_schema_uri\": schema}]\n", + " print(\"dataset:\", dataset_id)\n", + " start_time = time.time()\n", + " try:\n", + " operation = clients[\"dataset\"].import_data(\n", + " name=dataset_id, import_configs=config\n", + " )\n", + " print(\"Long running operation:\", operation.operation.name)\n", + "\n", + " result = operation.result()\n", + " print(\"result:\", result)\n", + " print(\"time:\", int(time.time() - start_time), \"secs\")\n", + " print(\"error:\", operation.exception())\n", + " print(\"meta :\", operation.metadata)\n", + " print(\n", + " \"after: running:\",\n", + " operation.running(),\n", + " \"done:\",\n", + " operation.done(),\n", + " \"cancelled:\",\n", + " operation.cancelled(),\n", + " )\n", + "\n", + " return operation\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "import_data(dataset_id, [IMPORT_FILE], LABEL_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "export_dataset" + }, + "source": [ + "### Export dataset index\n", + "\n", + "Next, you will export the dataset index to a JSONL file which will then be used by your custom training job to get the data and corresponding labels for training your Iris model. Use this helper function `export_data` to export the dataset index. The function does the following:\n", + "\n", + "- Uses the dataset client.\n", + "- Calls the client method `export_data`, with the following parameters:\n", + " - `name`: The human readable name you give to the dataset (e.g., iris).\n", + " - `export_config`: The export configuration.\n", + "- `export_config` A python list containing a dictionary, with the key/value entries:\n", + " - `gcs_destination`: The Cloud Storage bucket to write the JSONL dataset index file to.\n", + "\n", + "The `export_data()` method returns a long running `operation` object. This will take a few minutes to complete. The helper function will return the long running operation and the result of the operation when the export has completed." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "export_dataset" + }, + "outputs": [], + "source": [ + "EXPORT_FILE = BUCKET_NAME + \"/export\"\n", + "\n", + "\n", + "def export_data(dataset_id, gcs_dest):\n", + " config = {\"gcs_destination\": {\"output_uri_prefix\": gcs_dest}}\n", + " start_time = time.time()\n", + " try:\n", + " operation = clients[\"dataset\"].export_data(\n", + " name=dataset_id, export_config=config\n", + " )\n", + " print(\"Long running operation:\", operation.operation.name)\n", + "\n", + " result = operation.result()\n", + " print(\"result:\", result)\n", + " print(\"time:\", int(time.time() - start_time), \"secs\")\n", + " print(\"error:\", operation.exception())\n", + " print(\"meta :\", operation.metadata)\n", + " print(\n", + " \"after: running:\",\n", + " operation.running(),\n", + " \"done:\",\n", + " operation.done(),\n", + " \"cancelled:\",\n", + " operation.cancelled(),\n", + " )\n", + "\n", + " return operation, result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None, None\n", + "\n", + "\n", + "_, result = export_data(dataset_id, EXPORT_FILE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_model" + }, + "source": [ + "## Train a model\n", + "\n", + "There are two ways you can train a custom model using a container image:\n", + "\n", + "- **Use a Google Cloud prebuilt container**. If you use a prebuilt container, you will additionally specify a Python package to install into the container image. This Python package contains your code for training a custom model.\n", + "\n", + "- **Use your own custom container image**. If you use your own container, the container needs to contain your code for training a custom model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_specification:prebuilt_container" + }, + "source": [ + "## Prepare your custom job specification\n", + "\n", + "Now that your clients are ready, your first step is to create a Job Specification for your custom training job. The job specification will consist of the following:\n", + "\n", + "- `worker_pool_spec` : The specification of the type of machine(s) you will use for training and how many (single or distributed)\n", + "- `python_package_spec` : The specification of the Python package to be installed with the pre-built container." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "source": [ + "### Prepare your machine specification\n", + "\n", + "Now define the machine specification for your custom training job. This tells Vertex what type of machine instance to provision for the training.\n", + " - `machine_type`: The type of GCP instance to provision -- e.g., n1-standard-8.\n", + " - `accelerator_type`: The type, if any, of hardware accelerator. In this tutorial if you previously set the variable `TRAIN_GPU != None`, you are using a GPU; otherwise you will use a CPU.\n", + " - `accelerator_count`: The number of accelerators." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "outputs": [], + "source": [ + "if TRAIN_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": TRAIN_COMPUTE,\n", + " \"accelerator_type\": TRAIN_GPU,\n", + " \"accelerator_count\": TRAIN_NGPU,\n", + " }\n", + "else:\n", + " machine_spec = {\"machine_type\": TRAIN_COMPUTE, \"accelerator_count\": 0}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "source": [ + "### Prepare your disk specification\n", + "\n", + "(optional) Now define the disk specification for your custom training job. This tells Vertex what type and size of disk to provision in each machine instance for the training.\n", + "\n", + " - `boot_disk_type`: Either SSD or Standard. SSD is faster, and Standard is less expensive. Defaults to SSD.\n", + " - `boot_disk_size_gb`: Size of disk in GB." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "outputs": [], + "source": [ + "DISK_TYPE = \"pd-ssd\" # [ pd-ssd, pd-standard]\n", + "DISK_SIZE = 200 # GB\n", + "\n", + "disk_spec = {\"boot_disk_type\": DISK_TYPE, \"boot_disk_size_gb\": DISK_SIZE}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container,tfrecord" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_job_\" + TIMESTAMP\n", + "MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, JOB_NAME)\n", + "\n", + "if not TRAIN_NGPU or TRAIN_NGPU < 2:\n", + " TRAIN_STRATEGY = \"single\"\n", + "else:\n", + " TRAIN_STRATEGY = \"mirror\"\n", + "\n", + "EPOCHS = 20\n", + "STEPS = 100\n", + "\n", + "PARAM_FILE = BUCKET_NAME + \"/params.txt\"\n", + "\n", + "DIRECT = True\n", + "if DIRECT:\n", + " CMDARGS = [\n", + " \"--model-dir=\" + MODEL_DIR,\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "else:\n", + " CMDARGS = [\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "\n", + "worker_pool_spec = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": machine_spec,\n", + " \"disk_spec\": disk_spec,\n", + " \"python_package_spec\": {\n", + " \"executor_image_uri\": TRAIN_IMAGE,\n", + " \"package_uris\": [BUCKET_NAME + \"/trainer_iris.tar.gz\"],\n", + " \"python_module\": \"trainer.task\",\n", + " \"args\": CMDARGS,\n", + " },\n", + " }\n", + "]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "assemble_custom_job_specification" + }, + "source": [ + "### Assemble a job specification\n", + "\n", + "Now assemble the complete description for the custom job specification:\n", + "\n", + "- `display_name`: The human readable name you assign to this custom job.\n", + "- `job_spec`: The specification for the custom job.\n", + " - `worker_pool_specs`: The specification for the machine VM instances.\n", + " - `base_output_directory`: This tells the service the Cloud Storage location where to save the model artifacts (when variable `DIRECT = False`). The service will then pass the location to the training script as the environment variable `AIP_MODEL_DIR`, and the path will be of the form:\n", + "\n", + " /model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "assemble_custom_job_specification" + }, + "outputs": [], + "source": [ + "if DIRECT:\n", + " job_spec = {\"worker_pool_specs\": worker_pool_spec}\n", + "else:\n", + " job_spec = {\n", + " \"worker_pool_specs\": worker_pool_spec,\n", + " \"base_output_directory\": {\"output_uri_prefix\": MODEL_DIR},\n", + " }\n", + "\n", + "custom_job = {\"display_name\": JOB_NAME, \"job_spec\": job_spec}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "examine_training_package" + }, + "source": [ + "### Examine the training package\n", + "\n", + "#### Package layout\n", + "\n", + "Before you start the training, you will look at how a Python package is assembled for a custom training job. When unarchived, the package contains the following directory/file layout.\n", + "\n", + "- PKG-INFO\n", + "- README.md\n", + "- setup.cfg\n", + "- setup.py\n", + "- trainer\n", + " - \\_\\_init\\_\\_.py\n", + " - task.py\n", + "\n", + "The files `setup.cfg` and `setup.py` are the instructions for installing the package into the operating environment of the Docker image.\n", + "\n", + "The file `trainer/task.py` is the Python script for executing the custom training job. *Note*, when we referred to it in the worker pool specification, we replace the directory slash with a dot (`trainer.task`) and dropped the file suffix (`.py`).\n", + "\n", + "#### Package Assembly\n", + "\n", + "In the following cells, you will assemble the training package." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "examine_training_package" + }, + "outputs": [], + "source": [ + "# Make folder for Python training script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\ntag_build =\\n\\ntag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'tensorflow_datasets==1.3.0',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\nName: Iris tabular classification\\n\\nVersion: 0.0.0\\n\\nSummary: Demostration training script\\n\\nHome-page: www.google.com\\n\\nAuthor: Google\\n\\nAuthor-email: aferlitsch@google.com\\n\\nLicense: Public\\n\\nDescription: Demo\\n\\nPlatform: Vertex\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tarball_training_script" + }, + "source": [ + "#### Store training script on your Cloud Storage bucket\n", + "\n", + "Next, you package the training folder into a compressed tar ball, and then store it in your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tarball_training_script" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_iris.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job" + }, + "source": [ + "### Train the model\n", + "\n", + "\n", + "Now start the training of your custom training job on Vertex. Use this helper function `create_custom_job`, which takes the following parameter:\n", + "\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "The helper function calls job client service's `create_custom_job` method, with the following parameters:\n", + "\n", + "-`parent`: The Vertex location path to `Dataset`, `Model` and `Endpoint` resources.\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "You will display a handful of the fields returned in `response` object, with the two that are of most interest are:\n", + "\n", + "`response.name`: The Vertex fully qualified identifier assigned to this custom training job. You save this identifier for using in subsequent steps.\n", + "\n", + "`response.state`: The current state of the custom training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job" + }, + "outputs": [], + "source": [ + "def create_custom_job(custom_job):\n", + " response = clients[\"job\"].create_custom_job(parent=PARENT, custom_job=custom_job)\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = create_custom_job(custom_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "job_id:response" + }, + "source": [ + "Now get the unique identifier for the custom job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom job\n", + "job_id = response.name\n", + "# The short numeric ID for the custom job\n", + "job_short_id = job_id.split(\"/\")[-1]\n", + "\n", + "print(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_custom_job" + }, + "source": [ + "### Get information on a custom job\n", + "\n", + "Next, use this helper function `get_custom_job`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "The helper function calls the job client service's`get_custom_job` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "If you recall, you got the Vertex fully qualified identifier for the custom job in the `response.name` field when you called the `create_custom_job` method, and saved the identifier in the variable `job_id`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_custom_job" + }, + "outputs": [], + "source": [ + "def get_custom_job(name, silent=False):\n", + " response = clients[\"job\"].get_custom_job(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = get_custom_job(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete:custom" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, we will need to know the location of the saved model, which the Python script saved in your local Cloud Storage bucket at `MODEL_DIR + '/saved_model.pb'`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete:custom" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_custom_job(job_id, True)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_path_to_deploy = None\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " if not DIRECT:\n", + " MODEL_DIR = MODEL_DIR + \"/model\"\n", + " model_path_to_deploy = MODEL_DIR\n", + " print(\"Training Time:\", response.update_time - response.create_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model_to_deploy:\", model_path_to_deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "load_saved_model" + }, + "source": [ + "## Load the saved model\n", + "\n", + "Your model is stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Now load it from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n", + "\n", + "To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "load_saved_model" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:image" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load the Iris test (holdout) data from `tf.keras.datasets`, using the method `load_data()`. This will return the dataset as a tuple of two elements. The first element is the training data and the second is the test data. Each element is also a tuple of two elements: the image data, and the corresponding labels.\n", + "\n", + "You don't need the training data, and hence why we loaded it as `(_, _)`.\n", + "\n", + "Before you can run the data through evaluation, you need to preprocess it:\n", + "\n", + "x_test:\n", + "1. Normalize (rescaling) the pixel data by dividing each pixel by 255. This will replace each single byte integer pixel with a 32-bit floating point number between 0 and 1.\n", + "\n", + "y_test:
\n", + "2. The labels are currently scalar (sparse). If you look back at the `compile()` step in the `trainer/task.py` script, you will find that it was compiled for sparse labels. So we don't need to do anything more." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:image,exported_ds" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now let's find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load some sample data from the end of the Iris dataset, last 10 items. We will then preprocessed the data items to form:\n", + "\n", + "- `x_test`: The preprocessed image data in memory.\n", + "- `y_test`: The corresponding labels." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "evaluate_custom_model:image,exported_ds" + }, + "outputs": [], + "source": [ + "x_test = []\n", + "y_test = []\n", + "\n", + "data_items = export_data_items[-10:]\n", + "for data_item in data_items:\n", + " data_item = json.loads(data_item)\n", + " print(\"FILE\", data_item[\"imageGcsUri\"])\n", + " with tf.io.gfile.GFile(data_item[\"imageGcsUri\"], \"rb\") as f:\n", + " data = f.read()\n", + " image = tf.io.decode_image(data)\n", + " image = tf.image.resize(image, (128, 128))\n", + " image = (image.numpy() / 255.0).astype(np.float32)\n", + " cls = data_item[\"classificationAnnotation\"][\"displayName\"]\n", + " label = cls2label[cls]\n", + " x_test.append(image)\n", + " y_test.append(label)\n", + "\n", + "x_test = np.asarray(x_test)\n", + "y_test = np.asarray(y_test)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "perform_evaluation_custom" + }, + "source": [ + "### Perform the model evaluation\n", + "\n", + "Now evaluate how well the model in the custom job did." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "perform_evaluation_custom" + }, + "outputs": [], + "source": [ + "model.evaluate(x_test, y_test)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "how_serving_function_works" + }, + "source": [ + "## Upload the model for serving\n", + "\n", + "Next, you will upload your TF.Keras model from the custom job to Vertex `Model` service, which will create a Vertex `Model` resource for your custom model. During upload, you need to define a serving function to convert data to the format your model expects. If you send encoded data to Vertex, your serving function ensures that the data is decoded on the model server before it is passed as input to your model.\n", + "\n", + "### How does the serving function work\n", + "\n", + "When you send a request to an online prediction server, the request is received by a HTTP server. The HTTP server extracts the prediction request from the HTTP request content body. The extracted prediction request is forwarded to the serving function. For Google pre-built prediction containers, the request content is passed to the serving function as a `tf.string`.\n", + "\n", + "The serving function consists of two parts:\n", + "\n", + "- `preprocessing function`:\n", + " - Converts the input (`tf.string`) to the input shape and data type of the underlying model (dynamic graph).\n", + " - Performs the same preprocessing of the data that was done during training the underlying model -- e.g., normalizing, scaling, etc.\n", + "- `post-processing function`:\n", + " - Converts the model output to format expected by the receiving application -- e.q., compresses the output.\n", + " - Packages the output for the the receiving application -- e.g., add headings, make JSON object, etc.\n", + "\n", + "Both the preprocessing and post-processing functions are converted to static graphs which are fused to the model. The output from the underlying model is passed to the post-processing function. The post-processing function passes the converted/packaged output back to the HTTP server. The HTTP server returns the output as the HTTP response content.\n", + "\n", + "One consideration you need to consider when building serving functions for TF.Keras models is that they run as static graphs. That means, you cannot use TF graph operations that require a dynamic graph. If you do, you will get an error during the compile of the serving function which will indicate that you are using an EagerTensor which is not supported." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_signature" + }, + "source": [ + "## Get the serving function signature\n", + "\n", + "You can get the signatures of your model's input and output layers by reloading the model into memory, and querying it for the signatures corresponding to each layer.\n", + "\n", + "When making a prediction request, you need to route the request to the serving function instead of the model, so you need to know the input layer name of the serving function -- which you will use later when you make a prediction request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_signature" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_path_to_deploy)\n", + "\n", + "serving_input = list(\n", + " loaded.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "print(\"Serving function input:\", serving_input)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "upload_the_model" + }, + "source": [ + "### Upload the model\n", + "\n", + "Use this helper function `upload_model` to upload your model, stored in SavedModel format, up to the `Model` service, which will instantiate a Vertex `Model` resource instance for your model. Once you've done that, you can use the `Model` resource instance in the same way as any other Vertex `Model` resource instance, such as deploying to an `Endpoint` resource for serving predictions.\n", + "\n", + "The helper function takes the following parameters:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` service.\n", + "- `image_uri`: The container image for the model deployment.\n", + "- `model_uri`: The Cloud Storage path to our SavedModel artificat. For this tutorial, this is the Cloud Storage location where the `trainer/task.py` saved the model artifacts, which we specified in the variable `MODEL_DIR`.\n", + "\n", + "The helper function calls the `Model` client service's method `upload_model`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for `Dataset`, `Model` and `Endpoint` resources.\n", + "- `model`: The specification for the Vertex `Model` resource instance.\n", + "\n", + "Let's now dive deeper into the Vertex model specification `model`. This is a dictionary object that consists of the following fields:\n", + "\n", + "- `display_name`: A human readable name for the `Model` resource.\n", + "- `metadata_schema_uri`: Since your model was built without an Vertex `Dataset` resource, you will leave this blank (`''`).\n", + "- `artificat_uri`: The Cloud Storage path where the model is stored in SavedModel format.\n", + "- `container_spec`: This is the specification for the Docker container that will be installed on the `Endpoint` resource, from which the `Model` resource will serve predictions. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + "\n", + "Uploading a model into a Vertex Model resource returns a long running operation, since it may take a few moments. You call response.result(), which is a synchronous call and will return when the Vertex Model resource is ready.\n", + "\n", + "The helper function returns the Vertex fully qualified identifier for the corresponding Vertex Model instance upload_model_response.model. You will save the identifier for subsequent steps in the variable model_to_deploy_id." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "upload_the_model" + }, + "outputs": [], + "source": [ + "IMAGE_URI = DEPLOY_IMAGE\n", + "\n", + "\n", + "def upload_model(display_name, image_uri, model_uri):\n", + " model = {\n", + " \"display_name\": display_name,\n", + " \"metadata_schema_uri\": \"\",\n", + " \"artifact_uri\": model_uri,\n", + " \"container_spec\": {\n", + " \"image_uri\": image_uri,\n", + " \"command\": [],\n", + " \"args\": [],\n", + " \"env\": [{\"name\": \"env_name\", \"value\": \"env_value\"}],\n", + " \"ports\": [{\"container_port\": 8080}],\n", + " \"predict_route\": \"\",\n", + " \"health_route\": \"\",\n", + " },\n", + " }\n", + " response = clients[\"model\"].upload_model(parent=PARENT, model=model)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " upload_model_response = response.result(timeout=180)\n", + " print(\"upload_model_response\")\n", + " print(\" model:\", upload_model_response.model)\n", + " return upload_model_response.model\n", + "\n", + "\n", + "model_to_deploy_id = upload_model(\"iris-\" + TIMESTAMP, IMAGE_URI, model_path_to_deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_model" + }, + "source": [ + "### Get `Model` resource information\n", + "\n", + "Now let's get the model information for just your model. Use this helper function `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource.\n", + "\n", + "This helper function calls the Vertex `Model` client service's method `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_model" + }, + "outputs": [], + "source": [ + "def get_model(name):\n", + " response = clients[\"model\"].get_model(name=name)\n", + " print(response)\n", + "\n", + "\n", + "get_model(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:custom" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex custom `Model` resource. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"iris_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:dedicated" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `dedicated_resources`: This refers to how many compute instances (replicas) that are scaled for serving prediction requests.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `min_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:dedicated" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"iris_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " \"machine_spec\": machine_spec,\n", + " },\n", + " \"disable_container_logging\": False,\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:exported_ds" + }, + "outputs": [], + "source": [ + "# Last data item in exported dataset index\n", + "data_items = export_data_items[-1:]\n", + "data_item = json.loads(data_items[0])\n", + "image_path = data_item[\"imageGcsUri\"]\n", + "print(\"IMAGE PATH\", image_path)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "prepare_test_item:exported_ds" + }, + "source": [ + "### Prepare the request content\n", + "You are going to send the image as compressed JPG image, instead of the raw uncompressed bytes:\n", + "\n", + "- `cv2.imread`: Read the compressed JPG images back into memory as raw bytes.\n", + "- `tf.image.resize`: Resize the image to the input shape of the model -- (128, 128, 3).\n", + "- cv2.imwrite: Write resized image back to disk.\n", + "- `base64.b64encode`: Read back the resized compressed image and encode into a base 64 encoded string." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "prepare_test_item:exported_ds" + }, + "outputs": [], + "source": [ + "! gsutil cp $image_path tmp.jpg\n", + "\n", + "import base64\n", + "\n", + "import cv2\n", + "\n", + "test_image = cv2.imread(\"tmp.jpg\", cv2.IMREAD_COLOR)\n", + "print(\"before:\", test_image.shape)\n", + "test_image = cv2.resize(test_image, (128, 128))\n", + "print(\"after:\", test_image.shape)\n", + "cv2.imwrite(\"tmp.jpg\", test_image.astype(np.uint8))\n", + "\n", + "# bytes = tf.io.read_file('tmp.jpg')\n", + "with open(\"tmp.jpg\", \"rb\") as f:\n", + " bytes = f.read()\n", + "b64str = base64.b64encode(bytes).decode(\"utf-8\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "send_prediction_request:tabular" + }, + "source": [ + "### Send the prediction request\n", + "\n", + "Ok, now you have a test data item. Use this helper function `predict_data`, which takes the parameters:\n", + "\n", + "- `data`: The test data item as a numpy 1D array of floating point values.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `parameters_dict`: Additional parameters for serving.\n", + "\n", + "This function uses the prediction client service and calls the `predict` method with the parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `instances`: A list of instances (data items) to predict.\n", + "- `parameters`: Additional parameters for serving.\n", + "\n", + "To pass the test data to the prediction service, you package it for transmission to the serving binary as follows:\n", + "\n", + " 1. Convert the data item from a 1D numpy array to a 1D Python list.\n", + " 2. Convert the prediction request to a serialized Google protobuf (`json_format.ParseDict()`)\n", + "\n", + "\n", + "Each instance in the prediction request is a dictionary entry of the form:\n", + "\n", + " {input_name: content}\n", + "\n", + "- `input_name`: the name of the input layer of the underlying model.\n", + "- `content`: The data item as a 1D Python list.\n", + "\n", + "Since the `predict()` service can take multiple data items (instances), you will send your single data item as a list of one data item. As a final step, you package the instances list into Google's protobuf format -- which is what we pass to the `predict()` service.\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding image in the request. You will see in the output for each prediction:\n", + "\n", + "- `predictions` -- the predicated median value of a house in units of 1K USD." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "send_prediction_request:tabular" + }, + "outputs": [], + "source": [ + "def predict_data(data, endpoint, parameters_dict):\n", + " parameters = json_format.ParseDict(parameters_dict, Value())\n", + "\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [{serving_input: data.tolist()}]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", prediction)\n", + "\n", + "\n", + "predict_data(test_item, endpoint_id, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_custom_tabular_classification_online_exported_ds.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_custom_tabular_regression_batch.ipynb b/notebooks/community/gapic/custom/showcase_custom_tabular_regression_batch.ipynb new file mode 100644 index 000000000..08e43de87 --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_custom_tabular_regression_batch.ipynb @@ -0,0 +1,2143 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Custom training tabular regression model for batch prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to train and deploy a custom tabular regression model for batch prediction." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:custom,boston,lrg" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Boston Housing Prices dataset](https://www.cs.toronto.edu/~delve/data/boston/bostonDetail.html). The version of the dataset you will use in this tutorial is built into TensorFlow. The trained model predicts the median price of a house in units of 1K USD." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,training,batch_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create a custom model, with a training pipeline, from a Python script in a Google prebuilt Docker container using the Vertex client library, and then do a batch prediction on the uploaded model. You can alternatively create custom models using `gcloud` command-line tool or online using Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex custom job for training a model.\n", + "- Train the TensorFlow model.\n", + "- Retrieve and load the model artifacts.\n", + "- View the model evaluation.\n", + "- Upload the model as a Vertex `Model` resource.\n", + "- Make a batch prediction." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for training and prediction.\n", + "\n", + "Set the variables `TRAIN_GPU/TRAIN_NGPU` and `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU.\n", + "\n", + "*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3 -- which is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n", + " TRAIN_GPU, TRAIN_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n", + " )\n", + "else:\n", + " TRAIN_GPU, TRAIN_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (None, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:training,prediction" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for training and prediction\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-3:latest`\n", + " - TensorFlow 2.4\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-4:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/training/xgboost-cpu.1-1`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/training/scikit-learn-cpu.0-23:latest`\n", + " - Pytorch\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-5:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-6:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-7:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for training](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers).\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-3:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-90:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-82:latest`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-23:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-22:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-20:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for prediction](https://cloud.google.com/vertex-ai/docs/predictions/pre-built-containers)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf2-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf2-cpu.{}\".format(TF)\n", + "else:\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "TRAIN_IMAGE = \"gcr.io/cloud-aiplatform/training/{}:latest\".format(TRAIN_VERSION)\n", + "DEPLOY_IMAGE = \"gcr.io/cloud-aiplatform/prediction/{}:latest\".format(DEPLOY_VERSION)\n", + "\n", + "print(\"Training:\", TRAIN_IMAGE, TRAIN_GPU, TRAIN_NGPU)\n", + "print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU, DEPLOY_NGPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:training,prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for training and prediction.\n", + "\n", + "- Set the variables `TRAIN_COMPUTE` and `DEPLOY_COMPUTE` to configure the compute resources for the VMs you will use for for training and prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: The following is not supported for training:*\n", + "\n", + " - `standard`: 2 vCPUs\n", + " - `highcpu`: 2, 4 and 8 vCPUs\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_TRAIN_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Train machine type\", TRAIN_COMPUTE)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own custom model and training for Boston Housing." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:custom" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Model Service for `Model` resources.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:custom" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"job\"] = create_job_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_model" + }, + "source": [ + "## Train a model\n", + "\n", + "There are two ways you can train a custom model using a container image:\n", + "\n", + "- **Use a Google Cloud prebuilt container**. If you use a prebuilt container, you will additionally specify a Python package to install into the container image. This Python package contains your code for training a custom model.\n", + "\n", + "- **Use your own custom container image**. If you use your own container, the container needs to contain your code for training a custom model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_specification:prebuilt_container" + }, + "source": [ + "## Prepare your custom job specification\n", + "\n", + "Now that your clients are ready, your first step is to create a Job Specification for your custom training job. The job specification will consist of the following:\n", + "\n", + "- `worker_pool_spec` : The specification of the type of machine(s) you will use for training and how many (single or distributed)\n", + "- `python_package_spec` : The specification of the Python package to be installed with the pre-built container." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "source": [ + "### Prepare your machine specification\n", + "\n", + "Now define the machine specification for your custom training job. This tells Vertex what type of machine instance to provision for the training.\n", + " - `machine_type`: The type of GCP instance to provision -- e.g., n1-standard-8.\n", + " - `accelerator_type`: The type, if any, of hardware accelerator. In this tutorial if you previously set the variable `TRAIN_GPU != None`, you are using a GPU; otherwise you will use a CPU.\n", + " - `accelerator_count`: The number of accelerators." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "outputs": [], + "source": [ + "if TRAIN_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": TRAIN_COMPUTE,\n", + " \"accelerator_type\": TRAIN_GPU,\n", + " \"accelerator_count\": TRAIN_NGPU,\n", + " }\n", + "else:\n", + " machine_spec = {\"machine_type\": TRAIN_COMPUTE, \"accelerator_count\": 0}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "source": [ + "### Prepare your disk specification\n", + "\n", + "(optional) Now define the disk specification for your custom training job. This tells Vertex what type and size of disk to provision in each machine instance for the training.\n", + "\n", + " - `boot_disk_type`: Either SSD or Standard. SSD is faster, and Standard is less expensive. Defaults to SSD.\n", + " - `boot_disk_size_gb`: Size of disk in GB." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "outputs": [], + "source": [ + "DISK_TYPE = \"pd-ssd\" # [ pd-ssd, pd-standard]\n", + "DISK_SIZE = 200 # GB\n", + "\n", + "disk_spec = {\"boot_disk_type\": DISK_TYPE, \"boot_disk_size_gb\": DISK_SIZE}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container,tabular" + }, + "source": [ + "### Define the worker pool specification\n", + "\n", + "Next, you define the worker pool specification for your custom training job. The worker pool specification will consist of the following:\n", + "\n", + "- `replica_count`: The number of instances to provision of this machine type.\n", + "- `machine_spec`: The hardware specification.\n", + "- `disk_spec` : (optional) The disk storage specification.\n", + "\n", + "- `python_package`: The Python training package to install on the VM instance(s) and which Python module to invoke, along with command line arguments for the Python module.\n", + "\n", + "Let's dive deeper now into the python package specification:\n", + "\n", + "-`executor_image_spec`: This is the docker image which is configured for your custom training job.\n", + "\n", + "-`package_uris`: This is a list of the locations (URIs) of your python training packages to install on the provisioned instance. The locations need to be in a Cloud Storage bucket. These can be either individual python files or a zip (archive) of an entire package. In the later case, the job service will unzip (unarchive) the contents into the docker image.\n", + "\n", + "-`python_module`: The Python module (script) to invoke for running the custom training job. In this example, you will be invoking `trainer.task.py` -- note that it was not neccessary to append the `.py` suffix.\n", + "\n", + "-`args`: The command line arguments to pass to the corresponding Pythom module. In this example, you will be setting:\n", + " - `\"--model-dir=\" + MODEL_DIR` : The Cloud Storage location where to store the model artifacts. There are two ways to tell the training script where to save the model artifacts:\n", + " - direct: You pass the Cloud Storage location as a command line argument to your training script (set variable `DIRECT = True`), or\n", + " - indirect: The service passes the Cloud Storage location as the environment variable `AIP_MODEL_DIR` to your training script (set variable `DIRECT = False`). In this case, you tell the service the model artifact location in the job specification.\n", + " - `\"--epochs=\" + EPOCHS`: The number of epochs for training.\n", + " - `\"--steps=\" + STEPS`: The number of steps (batches) per epoch.\n", + " - `\"--distribute=\" + TRAIN_STRATEGY\"` : The training distribution strategy to use for single or distributed training.\n", + " - `\"single\"`: single device.\n", + " - `\"mirror\"`: all GPU devices on a single compute instance.\n", + " - `\"multi\"`: all GPU devices on all compute instances.\n", + " - `\"--param-file=\" + PARAM_FILE`: The Cloud Storage location for storing feature normalization values." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container,tabular" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_job_\" + TIMESTAMP\n", + "MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, JOB_NAME)\n", + "\n", + "if not TRAIN_NGPU or TRAIN_NGPU < 2:\n", + " TRAIN_STRATEGY = \"single\"\n", + "else:\n", + " TRAIN_STRATEGY = \"mirror\"\n", + "\n", + "EPOCHS = 20\n", + "STEPS = 100\n", + "\n", + "PARAM_FILE = BUCKET_NAME + \"/params.txt\"\n", + "\n", + "DIRECT = True\n", + "if DIRECT:\n", + " CMDARGS = [\n", + " \"--model-dir=\" + MODEL_DIR,\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " \"--param-file=\" + PARAM_FILE,\n", + " ]\n", + "else:\n", + " CMDARGS = [\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " \"--param-file=\" + PARAM_FILE,\n", + " ]\n", + "\n", + "worker_pool_spec = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": machine_spec,\n", + " \"disk_spec\": disk_spec,\n", + " \"python_package_spec\": {\n", + " \"executor_image_uri\": TRAIN_IMAGE,\n", + " \"package_uris\": [BUCKET_NAME + \"/trainer_boston.tar.gz\"],\n", + " \"python_module\": \"trainer.task\",\n", + " \"args\": CMDARGS,\n", + " },\n", + " }\n", + "]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "assemble_custom_job_specification" + }, + "source": [ + "### Assemble a job specification\n", + "\n", + "Now assemble the complete description for the custom job specification:\n", + "\n", + "- `display_name`: The human readable name you assign to this custom job.\n", + "- `job_spec`: The specification for the custom job.\n", + " - `worker_pool_specs`: The specification for the machine VM instances.\n", + " - `base_output_directory`: This tells the service the Cloud Storage location where to save the model artifacts (when variable `DIRECT = False`). The service will then pass the location to the training script as the environment variable `AIP_MODEL_DIR`, and the path will be of the form:\n", + "\n", + " /model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "assemble_custom_job_specification" + }, + "outputs": [], + "source": [ + "if DIRECT:\n", + " job_spec = {\"worker_pool_specs\": worker_pool_spec}\n", + "else:\n", + " job_spec = {\n", + " \"worker_pool_specs\": worker_pool_spec,\n", + " \"base_output_directory\": {\"output_uri_prefix\": MODEL_DIR},\n", + " }\n", + "\n", + "custom_job = {\"display_name\": JOB_NAME, \"job_spec\": job_spec}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "examine_training_package" + }, + "source": [ + "### Examine the training package\n", + "\n", + "#### Package layout\n", + "\n", + "Before you start the training, you will look at how a Python package is assembled for a custom training job. When unarchived, the package contains the following directory/file layout.\n", + "\n", + "- PKG-INFO\n", + "- README.md\n", + "- setup.cfg\n", + "- setup.py\n", + "- trainer\n", + " - \\_\\_init\\_\\_.py\n", + " - task.py\n", + "\n", + "The files `setup.cfg` and `setup.py` are the instructions for installing the package into the operating environment of the Docker image.\n", + "\n", + "The file `trainer/task.py` is the Python script for executing the custom training job. *Note*, when we referred to it in the worker pool specification, we replace the directory slash with a dot (`trainer.task`) and dropped the file suffix (`.py`).\n", + "\n", + "#### Package Assembly\n", + "\n", + "In the following cells, you will assemble the training package." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "examine_training_package" + }, + "outputs": [], + "source": [ + "# Make folder for Python training script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\ntag_build =\\n\\ntag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'tensorflow_datasets==1.3.0',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\nName: Boston Housing tabular regression\\n\\nVersion: 0.0.0\\n\\nSummary: Demostration training script\\n\\nHome-page: www.google.com\\n\\nAuthor: Google\\n\\nAuthor-email: aferlitsch@google.com\\n\\nLicense: Public\\n\\nDescription: Demo\\n\\nPlatform: Vertex\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:boston" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the training script task.py. I won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n", + "- Loads Boston Housing dataset from TF.Keras builtin datasets\n", + "- Builds a simple deep neural network model using TF.Keras model API.\n", + "- Compiles the model (`compile()`).\n", + "- Sets a training distribution strategy according to the argument `args.distribute`.\n", + "- Trains the model (`fit()`) with epochs specified by `args.epochs`.\n", + "- Saves the trained model (`save(args.model_dir)`) to the specified model directory.\n", + "- Saves the maximum value for each feature `f.write(str(params))` to the specified parameters file." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:boston" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Single, Mirror and Multi-Machine Distributed Training for Boston Housing\n", + "\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "import numpy as np\n", + "import argparse\n", + "import os\n", + "import sys\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv('AIP_MODEL_DIR'), type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.001, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=20, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=100, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "parser.add_argument('--param-file', dest='param_file',\n", + " default='/tmp/param.txt', type=str,\n", + " help='Output file for parameters')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "\n", + "# Single Machine, single compute device\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "# Single Machine, multiple compute device\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "# Multiple Machine, multiple compute device\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "# Multi-worker configuration\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "\n", + "def make_dataset():\n", + "\n", + " # Scaling Boston Housing data features\n", + " def scale(feature):\n", + " max = np.max(feature)\n", + " feature = (feature / max).astype(np.float)\n", + " return feature, max\n", + "\n", + " (x_train, y_train), (x_test, y_test) = tf.keras.datasets.boston_housing.load_data(\n", + " path=\"boston_housing.npz\", test_split=0.2, seed=113\n", + " )\n", + " params = []\n", + " for _ in range(13):\n", + " x_train[_], max = scale(x_train[_])\n", + " x_test[_], _ = scale(x_test[_])\n", + " params.append(max)\n", + "\n", + " # store the normalization (max) value for each feature\n", + " with tf.io.gfile.GFile(args.param_file, 'w') as f:\n", + " f.write(str(params))\n", + " return (x_train, y_train), (x_test, y_test)\n", + "\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_dnn_model():\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Dense(128, activation='relu', input_shape=(13,)),\n", + " tf.keras.layers.Dense(128, activation='relu'),\n", + " tf.keras.layers.Dense(1, activation='linear')\n", + " ])\n", + " model.compile(\n", + " loss='mse',\n", + " optimizer=tf.keras.optimizers.RMSprop(learning_rate=args.lr))\n", + " return model\n", + "\n", + "NUM_WORKERS = strategy.num_replicas_in_sync\n", + "# Here the batch size scales up by number of workers since\n", + "# `tf.data.Dataset.batch` expects the global batch size.\n", + "BATCH_SIZE = 16\n", + "GLOBAL_BATCH_SIZE = BATCH_SIZE * NUM_WORKERS\n", + "\n", + "with strategy.scope():\n", + " # Creation of dataset, and model building/compiling need to be within\n", + " # `strategy.scope()`.\n", + " model = build_and_compile_dnn_model()\n", + "\n", + "# Train the model\n", + "(x_train, y_train), (x_test, y_test) = make_dataset()\n", + "model.fit(x_train, y_train, epochs=args.epochs, batch_size=GLOBAL_BATCH_SIZE)\n", + "model.save(args.model_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tarball_training_script" + }, + "source": [ + "#### Store training script on your Cloud Storage bucket\n", + "\n", + "Next, you package the training folder into a compressed tar ball, and then store it in your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tarball_training_script" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_boston.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job" + }, + "source": [ + "### Train the model\n", + "\n", + "\n", + "Now start the training of your custom training job on Vertex. Use this helper function `create_custom_job`, which takes the following parameter:\n", + "\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "The helper function calls job client service's `create_custom_job` method, with the following parameters:\n", + "\n", + "-`parent`: The Vertex location path to `Dataset`, `Model` and `Endpoint` resources.\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "You will display a handful of the fields returned in `response` object, with the two that are of most interest are:\n", + "\n", + "`response.name`: The Vertex fully qualified identifier assigned to this custom training job. You save this identifier for using in subsequent steps.\n", + "\n", + "`response.state`: The current state of the custom training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job" + }, + "outputs": [], + "source": [ + "def create_custom_job(custom_job):\n", + " response = clients[\"job\"].create_custom_job(parent=PARENT, custom_job=custom_job)\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = create_custom_job(custom_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "job_id:response" + }, + "source": [ + "Now get the unique identifier for the custom job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom job\n", + "job_id = response.name\n", + "# The short numeric ID for the custom job\n", + "job_short_id = job_id.split(\"/\")[-1]\n", + "\n", + "print(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_custom_job" + }, + "source": [ + "### Get information on a custom job\n", + "\n", + "Next, use this helper function `get_custom_job`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "The helper function calls the job client service's`get_custom_job` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "If you recall, you got the Vertex fully qualified identifier for the custom job in the `response.name` field when you called the `create_custom_job` method, and saved the identifier in the variable `job_id`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_custom_job" + }, + "outputs": [], + "source": [ + "def get_custom_job(name, silent=False):\n", + " response = clients[\"job\"].get_custom_job(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = get_custom_job(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete:custom" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, we will need to know the location of the saved model, which the Python script saved in your local Cloud Storage bucket at `MODEL_DIR + '/saved_model.pb'`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete:custom" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_custom_job(job_id, True)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_path_to_deploy = None\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " if not DIRECT:\n", + " MODEL_DIR = MODEL_DIR + \"/model\"\n", + " model_path_to_deploy = MODEL_DIR\n", + " print(\"Training Time:\", response.update_time - response.create_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model_to_deploy:\", model_path_to_deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "load_saved_model" + }, + "source": [ + "## Load the saved model\n", + "\n", + "Your model is stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Now load it from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n", + "\n", + "To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "load_saved_model" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:tabular" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now let's find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load the Boston Housing test (holdout) data from `tf.keras.datasets`, using the method `load_data()`. This will return the dataset as a tuple of two elements. The first element is the training data and the second is the test data. Each element is also a tuple of two elements: the feature data, and the corresponding labels (median value of owner-occupied home).\n", + "\n", + "You don't need the training data, and hence why we loaded it as `(_, _)`.\n", + "\n", + "Before you can run the data through evaluation, you need to preprocess it:\n", + "\n", + "x_test:\n", + "1. Normalize (rescaling) the data in each column by dividing each value by the maximum value of that column. This will replace each single value with a 32-bit floating point number between 0 and 1." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "evaluate_custom_model:tabular,boston" + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "from tensorflow.keras.datasets import boston_housing\n", + "\n", + "(_, _), (x_test, y_test) = boston_housing.load_data(\n", + " path=\"boston_housing.npz\", test_split=0.2, seed=113\n", + ")\n", + "\n", + "\n", + "def scale(feature):\n", + " max = np.max(feature)\n", + " feature = (feature / max).astype(np.float32)\n", + " return feature\n", + "\n", + "\n", + "# Let's save one data item that has not been scaled\n", + "x_test_notscaled = x_test[0:1].copy()\n", + "\n", + "for _ in range(13):\n", + " x_test[_] = scale(x_test[_])\n", + "x_test = x_test.astype(np.float32)\n", + "\n", + "print(x_test.shape, x_test.dtype, y_test.shape)\n", + "print(\"scaled\", x_test[0])\n", + "print(\"unscaled\", x_test_notscaled)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "perform_evaluation_custom" + }, + "source": [ + "### Perform the model evaluation\n", + "\n", + "Now evaluate how well the model in the custom job did." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "perform_evaluation_custom" + }, + "outputs": [], + "source": [ + "model.evaluate(x_test, y_test)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "how_serving_function_works" + }, + "source": [ + "## Upload the model for serving\n", + "\n", + "Next, you will upload your TF.Keras model from the custom job to Vertex `Model` service, which will create a Vertex `Model` resource for your custom model. During upload, you need to define a serving function to convert data to the format your model expects. If you send encoded data to Vertex, your serving function ensures that the data is decoded on the model server before it is passed as input to your model.\n", + "\n", + "### How does the serving function work\n", + "\n", + "When you send a request to an online prediction server, the request is received by a HTTP server. The HTTP server extracts the prediction request from the HTTP request content body. The extracted prediction request is forwarded to the serving function. For Google pre-built prediction containers, the request content is passed to the serving function as a `tf.string`.\n", + "\n", + "The serving function consists of two parts:\n", + "\n", + "- `preprocessing function`:\n", + " - Converts the input (`tf.string`) to the input shape and data type of the underlying model (dynamic graph).\n", + " - Performs the same preprocessing of the data that was done during training the underlying model -- e.g., normalizing, scaling, etc.\n", + "- `post-processing function`:\n", + " - Converts the model output to format expected by the receiving application -- e.q., compresses the output.\n", + " - Packages the output for the the receiving application -- e.g., add headings, make JSON object, etc.\n", + "\n", + "Both the preprocessing and post-processing functions are converted to static graphs which are fused to the model. The output from the underlying model is passed to the post-processing function. The post-processing function passes the converted/packaged output back to the HTTP server. The HTTP server returns the output as the HTTP response content.\n", + "\n", + "One consideration you need to consider when building serving functions for TF.Keras models is that they run as static graphs. That means, you cannot use TF graph operations that require a dynamic graph. If you do, you will get an error during the compile of the serving function which will indicate that you are using an EagerTensor which is not supported." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_signature" + }, + "source": [ + "## Get the serving function signature\n", + "\n", + "You can get the signatures of your model's input and output layers by reloading the model into memory, and querying it for the signatures corresponding to each layer.\n", + "\n", + "When making a prediction request, you need to route the request to the serving function instead of the model, so you need to know the input layer name of the serving function -- which you will use later when you make a prediction request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_signature" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_path_to_deploy)\n", + "\n", + "serving_input = list(\n", + " loaded.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "print(\"Serving function input:\", serving_input)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "upload_the_model" + }, + "source": [ + "### Upload the model\n", + "\n", + "Use this helper function `upload_model` to upload your model, stored in SavedModel format, up to the `Model` service, which will instantiate a Vertex `Model` resource instance for your model. Once you've done that, you can use the `Model` resource instance in the same way as any other Vertex `Model` resource instance, such as deploying to an `Endpoint` resource for serving predictions.\n", + "\n", + "The helper function takes the following parameters:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` service.\n", + "- `image_uri`: The container image for the model deployment.\n", + "- `model_uri`: The Cloud Storage path to our SavedModel artificat. For this tutorial, this is the Cloud Storage location where the `trainer/task.py` saved the model artifacts, which we specified in the variable `MODEL_DIR`.\n", + "\n", + "The helper function calls the `Model` client service's method `upload_model`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for `Dataset`, `Model` and `Endpoint` resources.\n", + "- `model`: The specification for the Vertex `Model` resource instance.\n", + "\n", + "Let's now dive deeper into the Vertex model specification `model`. This is a dictionary object that consists of the following fields:\n", + "\n", + "- `display_name`: A human readable name for the `Model` resource.\n", + "- `metadata_schema_uri`: Since your model was built without an Vertex `Dataset` resource, you will leave this blank (`''`).\n", + "- `artificat_uri`: The Cloud Storage path where the model is stored in SavedModel format.\n", + "- `container_spec`: This is the specification for the Docker container that will be installed on the `Endpoint` resource, from which the `Model` resource will serve predictions. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + "\n", + "Uploading a model into a Vertex Model resource returns a long running operation, since it may take a few moments. You call response.result(), which is a synchronous call and will return when the Vertex Model resource is ready.\n", + "\n", + "The helper function returns the Vertex fully qualified identifier for the corresponding Vertex Model instance upload_model_response.model. You will save the identifier for subsequent steps in the variable model_to_deploy_id." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "upload_the_model" + }, + "outputs": [], + "source": [ + "IMAGE_URI = DEPLOY_IMAGE\n", + "\n", + "\n", + "def upload_model(display_name, image_uri, model_uri):\n", + " model = {\n", + " \"display_name\": display_name,\n", + " \"metadata_schema_uri\": \"\",\n", + " \"artifact_uri\": model_uri,\n", + " \"container_spec\": {\n", + " \"image_uri\": image_uri,\n", + " \"command\": [],\n", + " \"args\": [],\n", + " \"env\": [{\"name\": \"env_name\", \"value\": \"env_value\"}],\n", + " \"ports\": [{\"container_port\": 8080}],\n", + " \"predict_route\": \"\",\n", + " \"health_route\": \"\",\n", + " },\n", + " }\n", + " response = clients[\"model\"].upload_model(parent=PARENT, model=model)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " upload_model_response = response.result(timeout=180)\n", + " print(\"upload_model_response\")\n", + " print(\" model:\", upload_model_response.model)\n", + " return upload_model_response.model\n", + "\n", + "\n", + "model_to_deploy_id = upload_model(\n", + " \"boston-\" + TIMESTAMP, IMAGE_URI, model_path_to_deploy\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_model" + }, + "source": [ + "### Get `Model` resource information\n", + "\n", + "Now let's get the model information for just your model. Use this helper function `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource.\n", + "\n", + "This helper function calls the Vertex `Model` client service's method `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_model" + }, + "outputs": [], + "source": [ + "def get_model(name):\n", + " response = clients[\"model\"].get_model(name=name)\n", + " print(response)\n", + "\n", + "\n", + "get_model(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy:batch_prediction" + }, + "source": [ + "## Model deployment for batch prediction\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created for batch prediction. This differs from deploying a `Model` resource for on-demand prediction.\n", + "\n", + "For online prediction, you:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource.\n", + "\n", + "3. Make online prediction requests to the `Endpoint` resource.\n", + "\n", + "For batch-prediction, you:\n", + "\n", + "1. Create a batch prediction job.\n", + "\n", + "2. The job service will provision resources for the batch prediction request.\n", + "\n", + "3. The results of the batch prediction request are returned to the caller.\n", + "\n", + "4. The job service will unprovision the resoures for the batch prediction request." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a batch prediction request\n", + "\n", + "Now do a batch prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_items:test" + }, + "source": [ + "### Get test items\n", + "\n", + "You will use examples out of the test (holdout) portion of the dataset as a test items." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_items:test,tabular" + }, + "outputs": [], + "source": [ + "test_item_1 = x_test[0]\n", + "test_label_1 = y_test[0]\n", + "test_item_2 = x_test[1]\n", + "test_label_2 = y_test[1]\n", + "print(test_item_1.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:custom,tabular" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "Now make a batch input file, which you will store in your local Cloud Storage bucket. Each instance in the prediction request is a dictionary entry of the form:\n", + "\n", + " {serving_input: content}\n", + "\n", + "- `serving_input`: the name of the input layer of the underlying model.\n", + "- `content`: The feature values of the test item as a list." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_file:custom,tabular" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "gcs_input_uri = BUCKET_NAME + \"/\" + \"test.jsonl\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " data = {serving_input: test_item_1.tolist()}\n", + " f.write(json.dumps(data) + \"\\n\")\n", + " data = {serving_input: test_item_2.tolist()}\n", + " f.write(json.dumps(data) + \"\\n\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your batch prediction requests:\n", + "\n", + "- Single Instance: The batch prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The batch prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and batch prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The batch prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_request:custom" + }, + "source": [ + "### Make batch prediction request\n", + "\n", + "Now that your batch of two test items is ready, let's do the batch request. Use this helper function `create_batch_prediction_job`, with the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the prediction job.\n", + "- `model_name`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `gcs_source_uri`: The Cloud Storage path to the input file -- which you created above.\n", + "- `gcs_destination_output_uri_prefix`: The Cloud Storage path that the service will write the predictions to.\n", + "- `parameters`: Additional filtering parameters for serving prediction results.\n", + "\n", + "The helper function calls the job client service's `create_batch_prediction_job` metho, with the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for Dataset, Model and Pipeline resources.\n", + "- `batch_prediction_job`: The specification for the batch prediction job.\n", + "\n", + "Let's now dive into the specification for the `batch_prediction_job`:\n", + "\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `dedicated_resources`: The compute resources to provision for the batch prediction job.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `starting_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "- `model_parameters`: Additional filtering parameters for serving prediction results. No Additional parameters are supported for custom models.\n", + "- `input_config`: The input source and format type for the instances to predict.\n", + " - `instances_format`: The format of the batch prediction request file: `csv` or `jsonl`.\n", + " - `gcs_source`: A list of one or more Cloud Storage paths to your batch prediction requests.\n", + "- `output_config`: The output destination and format for the predictions.\n", + " - `prediction_format`: The format of the batch prediction response file: `csv` or `jsonl`.\n", + " - `gcs_destination`: The output destination for the predictions.\n", + "\n", + "This call is an asychronous operation. You will print from the response object a few select fields, including:\n", + "\n", + "- `name`: The Vertex fully qualified identifier assigned to the batch prediction job.\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the Model resource.\n", + "- `generate_explanations`: Whether True/False explanations were provided with the predictions (explainability).\n", + "- `state`: The state of the prediction job (pending, running, etc).\n", + "\n", + "Since this call will take a few moments to execute, you will likely get `JobState.JOB_STATE_PENDING` for `state`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_request:custom" + }, + "outputs": [], + "source": [ + "BATCH_MODEL = \"boston_batch-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_batch_prediction_job(\n", + " display_name,\n", + " model_name,\n", + " gcs_source_uri,\n", + " gcs_destination_output_uri_prefix,\n", + " parameters=None,\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " batch_prediction_job = {\n", + " \"display_name\": display_name,\n", + " # Format: 'projects/{project}/locations/{location}/models/{model_id}'\n", + " \"model\": model_name,\n", + " \"model_parameters\": json_format.ParseDict(parameters, Value()),\n", + " \"input_config\": {\n", + " \"instances_format\": IN_FORMAT,\n", + " \"gcs_source\": {\"uris\": [gcs_source_uri]},\n", + " },\n", + " \"output_config\": {\n", + " \"predictions_format\": OUT_FORMAT,\n", + " \"gcs_destination\": {\"output_uri_prefix\": gcs_destination_output_uri_prefix},\n", + " },\n", + " \"dedicated_resources\": {\n", + " \"machine_spec\": machine_spec,\n", + " \"starting_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " },\n", + " }\n", + " response = clients[\"job\"].create_batch_prediction_job(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + " )\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try:\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", response.labels)\n", + " return response\n", + "\n", + "\n", + "IN_FORMAT = \"jsonl\"\n", + "OUT_FORMAT = \"jsonl\"\n", + "\n", + "response = create_batch_prediction_job(\n", + " BATCH_MODEL, model_to_deploy_id, gcs_input_uri, BUCKET_NAME\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batch_job_id:response" + }, + "source": [ + "Now get the unique identifier for the batch prediction job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batch_job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the batch job\n", + "batch_job_id = response.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split(\"/\")[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_batch_prediction_job" + }, + "source": [ + "### Get information on a batch prediction job\n", + "\n", + "Use this helper function `get_batch_prediction_job`, with the following paramter:\n", + "\n", + "- `job_name`: The Vertex fully qualified identifier for the batch prediction job.\n", + "\n", + "The helper function calls the job client service's `get_batch_prediction_job` method, with the following paramter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the batch prediction job. In this tutorial, you will pass it the Vertex fully qualified identifier for your batch prediction job -- `batch_job_id`\n", + "\n", + "The helper function will return the Cloud Storage path to where the predictions are stored -- `gcs_destination`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_batch_prediction_job" + }, + "outputs": [], + "source": [ + "def get_batch_prediction_job(job_name, silent=False):\n", + " response = clients[\"job\"].get_batch_prediction_job(name=job_name)\n", + " if silent:\n", + " return response.output_config.gcs_destination.output_uri_prefix, response.state\n", + "\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try: # not all data types support explanations\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" error:\", response.error)\n", + " gcs_destination = response.output_config.gcs_destination\n", + " print(\" gcs_destination\")\n", + " print(\" output_uri_prefix:\", gcs_destination.output_uri_prefix)\n", + " return gcs_destination.output_uri_prefix, response.state\n", + "\n", + "\n", + "predictions, state = get_batch_prediction_job(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_the_predictions:custom,lrg" + }, + "source": [ + "### Get the predictions\n", + "\n", + "When the batch prediction is done processing, the job state will be `JOB_STATE_SUCCEEDED`.\n", + "\n", + "Finally you view the predictions stored at the Cloud Storage path you set as output. The predictions will be in a JSONL format, which you indicated at the time you made the batch prediction job, under a subfolder starting with the name `prediction`, and under that folder will be a file called `prediction.results-xxxxx-of-xxxxx`.\n", + "\n", + "Now display (cat) the contents. You will see multiple JSON objects, one for each prediction.\n", + "\n", + "The response contains a JSON object for each instance, in the form:\n", + "\n", + "- `dense_input`: The input for the prediction.\n", + "- `prediction`: The predicted value." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_the_predictions:custom,tabular" + }, + "outputs": [], + "source": [ + "def get_latest_predictions(gcs_out_dir):\n", + " \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n", + " folders = !gsutil ls $gcs_out_dir\n", + " latest = \"\"\n", + " for folder in folders:\n", + " subfolder = folder.split(\"/\")[-2]\n", + " if subfolder.startswith(\"prediction-\"):\n", + " if subfolder > latest:\n", + " latest = folder[:-1]\n", + " return latest\n", + "\n", + "\n", + "while True:\n", + " predictions, state = get_batch_prediction_job(batch_job_id, True)\n", + " if state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", state)\n", + " if state == aip.JobState.JOB_STATE_FAILED:\n", + " raise Exception(\"Batch Job Failed\")\n", + " else:\n", + " folder = get_latest_predictions(predictions)\n", + " ! gsutil ls $folder/prediction.results*\n", + "\n", + " print(\"Results:\")\n", + " ! gsutil cat $folder/prediction.results*\n", + "\n", + " print(\"Errors:\")\n", + " ! gsutil cat $folder/prediction.errors*\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_custom_tabular_regression_batch.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_custom_tabular_regression_batch_explain.ipynb b/notebooks/community/gapic/custom/showcase_custom_tabular_regression_batch_explain.ipynb new file mode 100644 index 000000000..4431f991e --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_custom_tabular_regression_batch_explain.ipynb @@ -0,0 +1,2342 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Custom training tabular regression model for batch prediction with explanation\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom,xai" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to train and deploy a custom tabular regression model for batch prediction with explanation." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:custom,boston,lrg" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Boston Housing Prices dataset](https://www.cs.toronto.edu/~delve/data/boston/bostonDetail.html). The version of the dataset you will use in this tutorial is built into TensorFlow. The trained model predicts the median price of a house in units of 1K USD." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,training,batch_prediction,xai" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create a custom model, with a training pipeline, from a Python script in a Google prebuilt Docker container using the Vertex client library, and then do a batch prediction with explanations on the uploaded model. You can alternatively create custom models using `gcloud` command-line tool or online using Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex custom job for training a model.\n", + "- Train the TensorFlow model.\n", + "- Retrieve and load the model artifacts.\n", + "- View the model evaluation.\n", + "- Set explanation parameters.\n", + "- Upload the model as a Vertex `Model` resource.\n", + "- Make a batch prediction with explanations." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_other_packages" + }, + "source": [ + "### Install other packages\n", + "\n", + "Install other packages required for this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_tabulate" + }, + "outputs": [], + "source": [ + "! pip3 install -U tabulate $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:v1beta1,protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:v1beta1,protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "import google.cloud.aiplatform_v1beta1 as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for training and prediction.\n", + "\n", + "Set the variables `TRAIN_GPU/TRAIN_NGPU` and `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU.\n", + "\n", + "*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3 -- which is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n", + " TRAIN_GPU, TRAIN_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n", + " )\n", + "else:\n", + " TRAIN_GPU, TRAIN_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (None, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:training,prediction" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for training and prediction\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-3:latest`\n", + " - TensorFlow 2.4\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-4:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/training/xgboost-cpu.1-1`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/training/scikit-learn-cpu.0-23:latest`\n", + " - Pytorch\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-5:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-6:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-7:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for training](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers).\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-3:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-90:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-82:latest`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-23:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-22:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-20:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for prediction](https://cloud.google.com/vertex-ai/docs/predictions/pre-built-containers)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf2-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf2-cpu.{}\".format(TF)\n", + "else:\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "TRAIN_IMAGE = \"gcr.io/cloud-aiplatform/training/{}:latest\".format(TRAIN_VERSION)\n", + "DEPLOY_IMAGE = \"gcr.io/cloud-aiplatform/prediction/{}:latest\".format(DEPLOY_VERSION)\n", + "\n", + "print(\"Training:\", TRAIN_IMAGE, TRAIN_GPU, TRAIN_NGPU)\n", + "print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU, DEPLOY_NGPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:training,prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for training and prediction.\n", + "\n", + "- Set the variables `TRAIN_COMPUTE` and `DEPLOY_COMPUTE` to configure the compute resources for the VMs you will use for for training and prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: The following is not supported for training:*\n", + "\n", + " - `standard`: 2 vCPUs\n", + " - `highcpu`: 2, 4 and 8 vCPUs\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_TRAIN_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Train machine type\", TRAIN_COMPUTE)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own custom model and training for Boston Housing." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:custom" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Model Service for `Model` resources.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:custom" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"job\"] = create_job_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_model" + }, + "source": [ + "## Train a model\n", + "\n", + "There are two ways you can train a custom model using a container image:\n", + "\n", + "- **Use a Google Cloud prebuilt container**. If you use a prebuilt container, you will additionally specify a Python package to install into the container image. This Python package contains your code for training a custom model.\n", + "\n", + "- **Use your own custom container image**. If you use your own container, the container needs to contain your code for training a custom model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_specification:prebuilt_container" + }, + "source": [ + "## Prepare your custom job specification\n", + "\n", + "Now that your clients are ready, your first step is to create a Job Specification for your custom training job. The job specification will consist of the following:\n", + "\n", + "- `worker_pool_spec` : The specification of the type of machine(s) you will use for training and how many (single or distributed)\n", + "- `python_package_spec` : The specification of the Python package to be installed with the pre-built container." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "source": [ + "### Prepare your machine specification\n", + "\n", + "Now define the machine specification for your custom training job. This tells Vertex what type of machine instance to provision for the training.\n", + " - `machine_type`: The type of GCP instance to provision -- e.g., n1-standard-8.\n", + " - `accelerator_type`: The type, if any, of hardware accelerator. In this tutorial if you previously set the variable `TRAIN_GPU != None`, you are using a GPU; otherwise you will use a CPU.\n", + " - `accelerator_count`: The number of accelerators." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "outputs": [], + "source": [ + "if TRAIN_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": TRAIN_COMPUTE,\n", + " \"accelerator_type\": TRAIN_GPU,\n", + " \"accelerator_count\": TRAIN_NGPU,\n", + " }\n", + "else:\n", + " machine_spec = {\"machine_type\": TRAIN_COMPUTE, \"accelerator_count\": 0}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "source": [ + "### Prepare your disk specification\n", + "\n", + "(optional) Now define the disk specification for your custom training job. This tells Vertex what type and size of disk to provision in each machine instance for the training.\n", + "\n", + " - `boot_disk_type`: Either SSD or Standard. SSD is faster, and Standard is less expensive. Defaults to SSD.\n", + " - `boot_disk_size_gb`: Size of disk in GB." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "outputs": [], + "source": [ + "DISK_TYPE = \"pd-ssd\" # [ pd-ssd, pd-standard]\n", + "DISK_SIZE = 200 # GB\n", + "\n", + "disk_spec = {\"boot_disk_type\": DISK_TYPE, \"boot_disk_size_gb\": DISK_SIZE}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container" + }, + "source": [ + "### Define the worker pool specification\n", + "\n", + "Next, you define the worker pool specification for your custom training job. The worker pool specification will consist of the following:\n", + "\n", + "- `replica_count`: The number of instances to provision of this machine type.\n", + "- `machine_spec`: The hardware specification.\n", + "- `disk_spec` : (optional) The disk storage specification.\n", + "\n", + "- `python_package`: The Python training package to install on the VM instance(s) and which Python module to invoke, along with command line arguments for the Python module.\n", + "\n", + "Let's dive deeper now into the python package specification:\n", + "\n", + "-`executor_image_spec`: This is the docker image which is configured for your custom training job.\n", + "\n", + "-`package_uris`: This is a list of the locations (URIs) of your python training packages to install on the provisioned instance. The locations need to be in a Cloud Storage bucket. These can be either individual python files or a zip (archive) of an entire package. In the later case, the job service will unzip (unarchive) the contents into the docker image.\n", + "\n", + "-`python_module`: The Python module (script) to invoke for running the custom training job. In this example, you will be invoking `trainer.task.py` -- note that it was not neccessary to append the `.py` suffix.\n", + "\n", + "-`args`: The command line arguments to pass to the corresponding Pythom module. In this example, you will be setting:\n", + " - `\"--model-dir=\" + MODEL_DIR` : The Cloud Storage location where to store the model artifacts. There are two ways to tell the training script where to save the model artifacts:\n", + " - direct: You pass the Cloud Storage location as a command line argument to your training script (set variable `DIRECT = True`), or\n", + " - indirect: The service passes the Cloud Storage location as the environment variable `AIP_MODEL_DIR` to your training script (set variable `DIRECT = False`). In this case, you tell the service the model artifact location in the job specification.\n", + " - `\"--epochs=\" + EPOCHS`: The number of epochs for training.\n", + " - `\"--steps=\" + STEPS`: The number of steps (batches) per epoch.\n", + " - `\"--distribute=\" + TRAIN_STRATEGY\"` : The training distribution strategy to use for single or distributed training.\n", + " - `\"single\"`: single device.\n", + " - `\"mirror\"`: all GPU devices on a single compute instance.\n", + " - `\"multi\"`: all GPU devices on all compute instances." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_job_\" + TIMESTAMP\n", + "MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, JOB_NAME)\n", + "\n", + "if not TRAIN_NGPU or TRAIN_NGPU < 2:\n", + " TRAIN_STRATEGY = \"single\"\n", + "else:\n", + " TRAIN_STRATEGY = \"mirror\"\n", + "\n", + "EPOCHS = 20\n", + "STEPS = 100\n", + "\n", + "DIRECT = True\n", + "if DIRECT:\n", + " CMDARGS = [\n", + " \"--model-dir=\" + MODEL_DIR,\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "else:\n", + " CMDARGS = [\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "\n", + "worker_pool_spec = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": machine_spec,\n", + " \"disk_spec\": disk_spec,\n", + " \"python_package_spec\": {\n", + " \"executor_image_uri\": TRAIN_IMAGE,\n", + " \"package_uris\": [BUCKET_NAME + \"/trainer_boston.tar.gz\"],\n", + " \"python_module\": \"trainer.task\",\n", + " \"args\": CMDARGS,\n", + " },\n", + " }\n", + "]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "assemble_custom_job_specification" + }, + "source": [ + "### Assemble a job specification\n", + "\n", + "Now assemble the complete description for the custom job specification:\n", + "\n", + "- `display_name`: The human readable name you assign to this custom job.\n", + "- `job_spec`: The specification for the custom job.\n", + " - `worker_pool_specs`: The specification for the machine VM instances.\n", + " - `base_output_directory`: This tells the service the Cloud Storage location where to save the model artifacts (when variable `DIRECT = False`). The service will then pass the location to the training script as the environment variable `AIP_MODEL_DIR`, and the path will be of the form:\n", + "\n", + " /model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "assemble_custom_job_specification" + }, + "outputs": [], + "source": [ + "if DIRECT:\n", + " job_spec = {\"worker_pool_specs\": worker_pool_spec}\n", + "else:\n", + " job_spec = {\n", + " \"worker_pool_specs\": worker_pool_spec,\n", + " \"base_output_directory\": {\"output_uri_prefix\": MODEL_DIR},\n", + " }\n", + "\n", + "custom_job = {\"display_name\": JOB_NAME, \"job_spec\": job_spec}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "examine_training_package" + }, + "source": [ + "### Examine the training package\n", + "\n", + "#### Package layout\n", + "\n", + "Before you start the training, you will look at how a Python package is assembled for a custom training job. When unarchived, the package contains the following directory/file layout.\n", + "\n", + "- PKG-INFO\n", + "- README.md\n", + "- setup.cfg\n", + "- setup.py\n", + "- trainer\n", + " - \\_\\_init\\_\\_.py\n", + " - task.py\n", + "\n", + "The files `setup.cfg` and `setup.py` are the instructions for installing the package into the operating environment of the Docker image.\n", + "\n", + "The file `trainer/task.py` is the Python script for executing the custom training job. *Note*, when we referred to it in the worker pool specification, we replace the directory slash with a dot (`trainer.task`) and dropped the file suffix (`.py`).\n", + "\n", + "#### Package Assembly\n", + "\n", + "In the following cells, you will assemble the training package." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "examine_training_package" + }, + "outputs": [], + "source": [ + "# Make folder for Python training script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\ntag_build =\\n\\ntag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'tensorflow_datasets==1.3.0',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\nName: Boston Housing tabular regression\\n\\nVersion: 0.0.0\\n\\nSummary: Demostration training script\\n\\nHome-page: www.google.com\\n\\nAuthor: Google\\n\\nAuthor-email: aferlitsch@google.com\\n\\nLicense: Public\\n\\nDescription: Demo\\n\\nPlatform: Vertex\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:boston" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the training script task.py. I won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n", + "- Loads Boston Housing dataset from TF.Keras builtin datasets\n", + "- Builds a simple deep neural network model using TF.Keras model API.\n", + "- Compiles the model (`compile()`).\n", + "- Sets a training distribution strategy according to the argument `args.distribute`.\n", + "- Trains the model (`fit()`) with epochs specified by `args.epochs`.\n", + "- Saves the trained model (`save(args.model_dir)`) to the specified model directory.\n", + "- Saves the maximum value for each feature `f.write(str(params))` to the specified parameters file." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:boston" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Single, Mirror and Multi-Machine Distributed Training for Boston Housing\n", + "\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "import numpy as np\n", + "import argparse\n", + "import os\n", + "import sys\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv('AIP_MODEL_DIR'), type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.001, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=20, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=100, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "parser.add_argument('--param-file', dest='param_file',\n", + " default='/tmp/param.txt', type=str,\n", + " help='Output file for parameters')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "\n", + "# Single Machine, single compute device\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "# Single Machine, multiple compute device\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "# Multiple Machine, multiple compute device\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "# Multi-worker configuration\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "\n", + "def make_dataset():\n", + "\n", + " # Scaling Boston Housing data features\n", + " def scale(feature):\n", + " max = np.max(feature)\n", + " feature = (feature / max).astype(np.float)\n", + " return feature, max\n", + "\n", + " (x_train, y_train), (x_test, y_test) = tf.keras.datasets.boston_housing.load_data(\n", + " path=\"boston_housing.npz\", test_split=0.2, seed=113\n", + " )\n", + " params = []\n", + " for _ in range(13):\n", + " x_train[_], max = scale(x_train[_])\n", + " x_test[_], _ = scale(x_test[_])\n", + " params.append(max)\n", + "\n", + " # store the normalization (max) value for each feature\n", + " with tf.io.gfile.GFile(args.param_file, 'w') as f:\n", + " f.write(str(params))\n", + " return (x_train, y_train), (x_test, y_test)\n", + "\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_dnn_model():\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Dense(128, activation='relu', input_shape=(13,)),\n", + " tf.keras.layers.Dense(128, activation='relu'),\n", + " tf.keras.layers.Dense(1, activation='linear')\n", + " ])\n", + " model.compile(\n", + " loss='mse',\n", + " optimizer=tf.keras.optimizers.RMSprop(learning_rate=args.lr))\n", + " return model\n", + "\n", + "NUM_WORKERS = strategy.num_replicas_in_sync\n", + "# Here the batch size scales up by number of workers since\n", + "# `tf.data.Dataset.batch` expects the global batch size.\n", + "BATCH_SIZE = 16\n", + "GLOBAL_BATCH_SIZE = BATCH_SIZE * NUM_WORKERS\n", + "\n", + "with strategy.scope():\n", + " # Creation of dataset, and model building/compiling need to be within\n", + " # `strategy.scope()`.\n", + " model = build_and_compile_dnn_model()\n", + "\n", + "# Train the model\n", + "(x_train, y_train), (x_test, y_test) = make_dataset()\n", + "model.fit(x_train, y_train, epochs=args.epochs, batch_size=GLOBAL_BATCH_SIZE)\n", + "model.save(args.model_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tarball_training_script" + }, + "source": [ + "#### Store training script on your Cloud Storage bucket\n", + "\n", + "Next, you package the training folder into a compressed tar ball, and then store it in your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tarball_training_script" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_boston.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job" + }, + "source": [ + "### Train the model\n", + "\n", + "\n", + "Now start the training of your custom training job on Vertex. Use this helper function `create_custom_job`, which takes the following parameter:\n", + "\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "The helper function calls job client service's `create_custom_job` method, with the following parameters:\n", + "\n", + "-`parent`: The Vertex location path to `Dataset`, `Model` and `Endpoint` resources.\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "You will display a handful of the fields returned in `response` object, with the two that are of most interest are:\n", + "\n", + "`response.name`: The Vertex fully qualified identifier assigned to this custom training job. You save this identifier for using in subsequent steps.\n", + "\n", + "`response.state`: The current state of the custom training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job" + }, + "outputs": [], + "source": [ + "def create_custom_job(custom_job):\n", + " response = clients[\"job\"].create_custom_job(parent=PARENT, custom_job=custom_job)\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = create_custom_job(custom_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "job_id:response" + }, + "source": [ + "Now get the unique identifier for the custom job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom job\n", + "job_id = response.name\n", + "# The short numeric ID for the custom job\n", + "job_short_id = job_id.split(\"/\")[-1]\n", + "\n", + "print(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_custom_job" + }, + "source": [ + "### Get information on a custom job\n", + "\n", + "Next, use this helper function `get_custom_job`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "The helper function calls the job client service's`get_custom_job` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "If you recall, you got the Vertex fully qualified identifier for the custom job in the `response.name` field when you called the `create_custom_job` method, and saved the identifier in the variable `job_id`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_custom_job" + }, + "outputs": [], + "source": [ + "def get_custom_job(name, silent=False):\n", + " response = clients[\"job\"].get_custom_job(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = get_custom_job(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete:custom" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, we will need to know the location of the saved model, which the Python script saved in your local Cloud Storage bucket at `MODEL_DIR + '/saved_model.pb'`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete:custom" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_custom_job(job_id, True)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_path_to_deploy = None\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " if not DIRECT:\n", + " MODEL_DIR = MODEL_DIR + \"/model\"\n", + " model_path_to_deploy = MODEL_DIR\n", + " print(\"Training Time:\", response.update_time - response.create_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model_to_deploy:\", model_path_to_deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "load_saved_model" + }, + "source": [ + "## Load the saved model\n", + "\n", + "Your model is stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Now load it from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n", + "\n", + "To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "load_saved_model" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:tabular" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now let's find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load the Boston Housing test (holdout) data from `tf.keras.datasets`, using the method `load_data()`. This will return the dataset as a tuple of two elements. The first element is the training data and the second is the test data. Each element is also a tuple of two elements: the feature data, and the corresponding labels (median value of owner-occupied home).\n", + "\n", + "You don't need the training data, and hence why we loaded it as `(_, _)`.\n", + "\n", + "Before you can run the data through evaluation, you need to preprocess it:\n", + "\n", + "x_test:\n", + "1. Normalize (rescaling) the data in each column by dividing each value by the maximum value of that column. This will replace each single value with a 32-bit floating point number between 0 and 1." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "evaluate_custom_model:tabular,boston" + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "from tensorflow.keras.datasets import boston_housing\n", + "\n", + "(_, _), (x_test, y_test) = boston_housing.load_data(\n", + " path=\"boston_housing.npz\", test_split=0.2, seed=113\n", + ")\n", + "\n", + "\n", + "def scale(feature):\n", + " max = np.max(feature)\n", + " feature = (feature / max).astype(np.float32)\n", + " return feature\n", + "\n", + "\n", + "# Let's save one data item that has not been scaled\n", + "x_test_notscaled = x_test[0:1].copy()\n", + "\n", + "for _ in range(13):\n", + " x_test[_] = scale(x_test[_])\n", + "x_test = x_test.astype(np.float32)\n", + "\n", + "print(x_test.shape, x_test.dtype, y_test.shape)\n", + "print(\"scaled\", x_test[0])\n", + "print(\"unscaled\", x_test_notscaled)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "perform_evaluation_custom" + }, + "source": [ + "### Perform the model evaluation\n", + "\n", + "Now evaluate how well the model in the custom job did." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "perform_evaluation_custom" + }, + "outputs": [], + "source": [ + "model.evaluate(x_test, y_test)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "how_serving_function_works" + }, + "source": [ + "## Upload the model for serving\n", + "\n", + "Next, you will upload your TF.Keras model from the custom job to Vertex `Model` service, which will create a Vertex `Model` resource for your custom model. During upload, you need to define a serving function to convert data to the format your model expects. If you send encoded data to Vertex, your serving function ensures that the data is decoded on the model server before it is passed as input to your model.\n", + "\n", + "### How does the serving function work\n", + "\n", + "When you send a request to an online prediction server, the request is received by a HTTP server. The HTTP server extracts the prediction request from the HTTP request content body. The extracted prediction request is forwarded to the serving function. For Google pre-built prediction containers, the request content is passed to the serving function as a `tf.string`.\n", + "\n", + "The serving function consists of two parts:\n", + "\n", + "- `preprocessing function`:\n", + " - Converts the input (`tf.string`) to the input shape and data type of the underlying model (dynamic graph).\n", + " - Performs the same preprocessing of the data that was done during training the underlying model -- e.g., normalizing, scaling, etc.\n", + "- `post-processing function`:\n", + " - Converts the model output to format expected by the receiving application -- e.q., compresses the output.\n", + " - Packages the output for the the receiving application -- e.g., add headings, make JSON object, etc.\n", + "\n", + "Both the preprocessing and post-processing functions are converted to static graphs which are fused to the model. The output from the underlying model is passed to the post-processing function. The post-processing function passes the converted/packaged output back to the HTTP server. The HTTP server returns the output as the HTTP response content.\n", + "\n", + "One consideration you need to consider when building serving functions for TF.Keras models is that they run as static graphs. That means, you cannot use TF graph operations that require a dynamic graph. If you do, you will get an error during the compile of the serving function which will indicate that you are using an EagerTensor which is not supported." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_signature:xai" + }, + "source": [ + "## Get the serving function signature\n", + "\n", + "You can get the signatures of your model's input and output layers by reloading the model into memory, and querying it for the signatures corresponding to each layer.\n", + "\n", + "When making a prediction request, you need to route the request to the serving function instead of the model, so you need to know the input layer name of the serving function -- which you will use later when you make a prediction request.\n", + "\n", + "You also need to know the name of the serving function's input and output layer for constructing the explanation metadata -- which is discussed subsequently." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_signature:xai" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_path_to_deploy)\n", + "\n", + "serving_input = list(\n", + " loaded.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "print(\"Serving function input:\", serving_input)\n", + "serving_output = list(loaded.signatures[\"serving_default\"].structured_outputs.keys())[0]\n", + "print(\"Serving function output:\", serving_output)\n", + "\n", + "input_name = model.input.name\n", + "print(\"Model input name:\", input_name)\n", + "output_name = model.output.name\n", + "print(\"Model output name:\", output_name)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "explanation_spec" + }, + "source": [ + "### Explanation Specification\n", + "\n", + "To get explanations when doing a prediction, you must enable the explanation capability and set corresponding settings when you upload your custom model to an Vertex `Model` resource. These settings are referred to as the explanation metadata, which consists of:\n", + "\n", + "- `parameters`: This is the specification for the explainability algorithm to use for explanations on your model. You can choose between:\n", + " - Shapley - *Note*, not recommended for image data -- can be very long running\n", + " - XRAI\n", + " - Integrated Gradients\n", + "- `metadata`: This is the specification for how the algoithm is applied on your custom model.\n", + "\n", + "#### Explanation Parameters\n", + "\n", + "Let's first dive deeper into the settings for the explainability algorithm.\n", + "\n", + "#### Shapley\n", + "\n", + "Assigns credit for the outcome to each feature, and considers different permutations of the features. This method provides a sampling approximation of exact Shapley values.\n", + "\n", + "Use Cases:\n", + " - Classification and regression on tabular data.\n", + "\n", + "Parameters:\n", + "\n", + "- `path_count`: This is the number of paths over the features that will be processed by the algorithm. An exact approximation of the Shapley values requires M! paths, where M is the number of features. For the CIFAR10 dataset, this would be 784 (28*28).\n", + "\n", + "For any non-trival number of features, this is too compute expensive. You can reduce the number of paths over the features to M * `path_count`.\n", + "\n", + "#### Integrated Gradients\n", + "\n", + "A gradients-based method to efficiently compute feature attributions with the same axiomatic properties as the Shapley value.\n", + "\n", + "Use Cases:\n", + " - Classification and regression on tabular data.\n", + " - Classification on image data.\n", + "\n", + "Parameters:\n", + "\n", + "- `step_count`: This is the number of steps to approximate the remaining sum. The more steps, the more accurate the integral approximation. The general rule of thumb is 50 steps, but as you increase so does the compute time.\n", + "\n", + "#### XRAI\n", + "\n", + "Based on the integrated gradients method, XRAI assesses overlapping regions of the image to create a saliency map, which highlights relevant regions of the image rather than pixels.\n", + "\n", + "Use Cases:\n", + "\n", + " - Classification on image data.\n", + "\n", + "Parameters:\n", + "\n", + "- `step_count`: This is the number of steps to approximate the remaining sum. The more steps, the more accurate the integral approximation. The general rule of thumb is 50 steps, but as you increase so does the compute time.\n", + "\n", + "In the next code cell, set the variable `XAI` to which explainabilty algorithm you will use on your custom model." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "explanation_spec" + }, + "outputs": [], + "source": [ + "XAI = \"ig\" # [ shapley, ig, xrai ]\n", + "\n", + "if XAI == \"shapley\":\n", + " PARAMETERS = {\"sampled_shapley_attribution\": {\"path_count\": 10}}\n", + "elif XAI == \"ig\":\n", + " PARAMETERS = {\"integrated_gradients_attribution\": {\"step_count\": 50}}\n", + "elif XAI == \"xrai\":\n", + " PARAMETERS = {\"xrai_attribution\": {\"step_count\": 50}}\n", + "\n", + "parameters = aip.ExplanationParameters(PARAMETERS)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "explanation_metadata:tabular" + }, + "source": [ + "#### Explanation Metadata\n", + "\n", + "Let's first dive deeper into the explanation metadata, which consists of:\n", + "\n", + "- `outputs`: A scalar value in the output to attribute -- what to explain. For example, in a probability output \\[0.1, 0.2, 0.7\\] for classification, one wants an explanation for 0.7. Consider the following formulae, where the output is `y` and that is what we want to explain.\n", + "\n", + " y = f(x)\n", + "\n", + "Consider the following formulae, where the outputs are `y` and `z`. Since we can only do attribution for one scalar value, we have to pick whether we want to explain the output `y` or `z`. Assume in this example the model is object detection and y and z are the bounding box and the object classification. You would want to pick which of the two outputs to explain.\n", + "\n", + " y, z = f(x)\n", + "\n", + "The dictionary format for `outputs` is:\n", + "\n", + " { \"outputs\": { \"[your_display_name]\":\n", + " \"output_tensor_name\": [layer]\n", + " }\n", + " }\n", + "\n", + "
\n", + " - [your_display_name]: A human readable name you assign to the output to explain. A common example is \"probability\".
\n", + " - \"output_tensor_name\": The key/value field to identify the output layer to explain.
\n", + " - [layer]: The output layer to explain. In a single task model, like a tabular regressor, it is the last (topmost) layer in the model.\n", + "
\n", + "\n", + "- `inputs`: The features for attribution -- how they contributed to the output. Consider the following formulae, where `a` and `b` are the features. We have to pick which features to explain how the contributed. Assume that this model is deployed for A/B testing, where `a` are the data_items for the prediction and `b` identifies whether the model instance is A or B. You would want to pick `a` (or some subset of) for the features, and not `b` since it does not contribute to the prediction.\n", + "\n", + " y = f(a,b)\n", + "\n", + "The minimum dictionary format for `inputs` is:\n", + "\n", + " { \"inputs\": { \"[your_display_name]\":\n", + " \"input_tensor_name\": [layer]\n", + " }\n", + " }\n", + "\n", + "
\n", + " - [your_display_name]: A human readable name you assign to the input to explain. A common example is \"features\".
\n", + " - \"input_tensor_name\": The key/value field to identify the input layer for the feature attribution.
\n", + " - [layer]: The input layer for feature attribution. In a single input tensor model, it is the first (bottom-most) layer in the model.\n", + "
\n", + "\n", + "Since the inputs to the model are tabular, you can specify the following two additional fields as reporting/visualization aids:\n", + "\n", + "
\n", + " - \"modality\": \"image\": Indicates the field values are image data.\n", + "
\n", + "\n", + "Since the inputs to the model are tabular, you can specify the following two additional fields as reporting/visualization aids:\n", + "\n", + "
\n", + " - \"encoding\": \"BAG_OF_FEATURES\" : Indicates that the inputs are set of tabular features.
\n", + " - \"index_feature_mapping\": [ feature-names ] : A list of human readable names for each feature. For this example, we use the feature names specified in the dataset.
\n", + " - \"modality\": \"numeric\": Indicates the field values are numeric.\n", + "
" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "explanation_metadata:tabular" + }, + "outputs": [], + "source": [ + "INPUT_METADATA = {\n", + " \"input_tensor_name\": serving_input,\n", + " \"encoding\": \"BAG_OF_FEATURES\",\n", + " \"modality\": \"numeric\",\n", + " \"index_feature_mapping\": [\n", + " \"crim\",\n", + " \"zn\",\n", + " \"indus\",\n", + " \"chas\",\n", + " \"nox\",\n", + " \"rm\",\n", + " \"age\",\n", + " \"dis\",\n", + " \"rad\",\n", + " \"tax\",\n", + " \"ptratio\",\n", + " \"b\",\n", + " \"lstat\",\n", + " ],\n", + "}\n", + "\n", + "OUTPUT_METADATA = {\"output_tensor_name\": serving_output}\n", + "\n", + "input_metadata = aip.ExplanationMetadata.InputMetadata(INPUT_METADATA)\n", + "output_metadata = aip.ExplanationMetadata.OutputMetadata(OUTPUT_METADATA)\n", + "\n", + "metadata = aip.ExplanationMetadata(\n", + " inputs={\"features\": input_metadata}, outputs={\"medv\": output_metadata}\n", + ")\n", + "\n", + "explanation_spec = aip.ExplanationSpec(metadata=metadata, parameters=parameters)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "upload_the_model:explanation" + }, + "source": [ + "### Upload the model\n", + "\n", + "Use this helper function `upload_model` to upload your model, stored in SavedModel format, up to the `Model` service, which will instantiate a Vertex `Model` resource instance for your model. Once you've done that, you can use the `Model` resource instance in the same way as any other Vertex `Model` resource instance, such as deploying to an `Endpoint` resource for serving predictions.\n", + "\n", + "The helper function takes the following parameters:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` service.\n", + "- `image_uri`: The container image for the model deployment.\n", + "- `model_uri`: The Cloud Storage path to our SavedModel artificat. For this tutorial, this is the Cloud Storage location where the `trainer/task.py` saved the model artifacts, which we specified in the variable `MODEL_DIR`.\n", + "\n", + "The helper function calls the `Model` client service's method `upload_model`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for `Dataset`, `Model` and `Endpoint` resources.\n", + "- `model`: The specification for the Vertex `Model` resource instance.\n", + "\n", + "Let's now dive deeper into the Vertex model specification `model`. This is a dictionary object that consists of the following fields:\n", + "\n", + "- `display_name`: A human readable name for the `Model` resource.\n", + "- `metadata_schema_uri`: Since your model was built without an Vertex `Dataset` resource, you will leave this blank (`''`).\n", + "- `artificat_uri`: The Cloud Storage path where the model is stored in SavedModel format.\n", + "- `container_spec`: This is the specification for the Docker container that will be installed on the `Endpoint` resource, from which the `Model` resource will serve predictions. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + "- `explanation_spec`: This is the specification for enabling explainability for your model.\n", + "\n", + "Uploading a model into a Vertex Model resource returns a long running operation, since it may take a few moments. You call response.result(), which is a synchronous call and will return when the Vertex Model resource is ready.\n", + "\n", + "The helper function returns the Vertex fully qualified identifier for the corresponding Vertex Model instance upload_model_response.model. You will save the identifier for subsequent steps in the variable model_to_deploy_id." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "upload_the_model:explanation" + }, + "outputs": [], + "source": [ + "IMAGE_URI = DEPLOY_IMAGE\n", + "\n", + "\n", + "def upload_model(display_name, image_uri, model_uri):\n", + "\n", + " model = aip.Model(\n", + " display_name=display_name,\n", + " artifact_uri=model_uri,\n", + " metadata_schema_uri=\"\",\n", + " explanation_spec=explanation_spec,\n", + " container_spec={\"image_uri\": image_uri},\n", + " )\n", + "\n", + " response = clients[\"model\"].upload_model(parent=PARENT, model=model)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " upload_model_response = response.result(timeout=180)\n", + " print(\"upload_model_response\")\n", + " print(\" model:\", upload_model_response.model)\n", + " return upload_model_response.model\n", + "\n", + "\n", + "model_to_deploy_id = upload_model(\n", + " \"boston-\" + TIMESTAMP, IMAGE_URI, model_path_to_deploy\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_model" + }, + "source": [ + "### Get `Model` resource information\n", + "\n", + "Now let's get the model information for just your model. Use this helper function `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource.\n", + "\n", + "This helper function calls the Vertex `Model` client service's method `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_model" + }, + "outputs": [], + "source": [ + "def get_model(name):\n", + " response = clients[\"model\"].get_model(name=name)\n", + " print(response)\n", + "\n", + "\n", + "get_model(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy:batch_prediction" + }, + "source": [ + "## Model deployment for batch prediction\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created for batch prediction. This differs from deploying a `Model` resource for on-demand prediction.\n", + "\n", + "For online prediction, you:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource.\n", + "\n", + "3. Make online prediction requests to the `Endpoint` resource.\n", + "\n", + "For batch-prediction, you:\n", + "\n", + "1. Create a batch prediction job.\n", + "\n", + "2. The job service will provision resources for the batch prediction request.\n", + "\n", + "3. The results of the batch prediction request are returned to the caller.\n", + "\n", + "4. The job service will unprovision the resoures for the batch prediction request." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a batch prediction request\n", + "\n", + "Now do a batch prediction to your deployed model." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_items:test,tabular" + }, + "outputs": [], + "source": [ + "test_item_1 = x_test[0]\n", + "test_label_1 = y_test[0]\n", + "test_item_2 = x_test[1]\n", + "test_label_2 = y_test[1]\n", + "print(test_item_1.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:custom,tabular" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "Now make a batch input file, which you will store in your local Cloud Storage bucket. Each instance in the prediction request is a dictionary entry of the form:\n", + "\n", + " {serving_input: content}\n", + "\n", + "- `serving_input`: the name of the input layer of the underlying model.\n", + "- `content`: The feature values of the test item as a list." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_file:custom,tabular" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "gcs_input_uri = BUCKET_NAME + \"/\" + \"test.jsonl\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " data = {serving_input: test_item_1.tolist()}\n", + " f.write(json.dumps(data) + \"\\n\")\n", + " data = {serving_input: test_item_2.tolist()}\n", + " f.write(json.dumps(data) + \"\\n\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your batch prediction requests:\n", + "\n", + "- Single Instance: The batch prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The batch prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and batch prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The batch prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_request:custom" + }, + "source": [ + "### Make batch prediction request\n", + "\n", + "Now that your batch of two test items is ready, let's do the batch request. Use this helper function `create_batch_prediction_job`, with the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the prediction job.\n", + "- `model_name`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `gcs_source_uri`: The Cloud Storage path to the input file -- which you created above.\n", + "- `gcs_destination_output_uri_prefix`: The Cloud Storage path that the service will write the predictions to.\n", + "- `parameters`: Additional filtering parameters for serving prediction results.\n", + "\n", + "The helper function calls the job client service's `create_batch_prediction_job` metho, with the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for Dataset, Model and Pipeline resources.\n", + "- `batch_prediction_job`: The specification for the batch prediction job.\n", + "\n", + "Let's now dive into the specification for the `batch_prediction_job`:\n", + "\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `dedicated_resources`: The compute resources to provision for the batch prediction job.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `starting_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "- `model_parameters`: Additional filtering parameters for serving prediction results. No Additional parameters are supported for custom models.\n", + "- `input_config`: The input source and format type for the instances to predict.\n", + " - `instances_format`: The format of the batch prediction request file: `csv` or `jsonl`.\n", + " - `gcs_source`: A list of one or more Cloud Storage paths to your batch prediction requests.\n", + "- `output_config`: The output destination and format for the predictions.\n", + " - `prediction_format`: The format of the batch prediction response file: `csv` or `jsonl`.\n", + " - `gcs_destination`: The output destination for the predictions.\n", + "\n", + "This call is an asychronous operation. You will print from the response object a few select fields, including:\n", + "\n", + "- `name`: The Vertex fully qualified identifier assigned to the batch prediction job.\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the Model resource.\n", + "- `generate_explanations`: Whether True/False explanations were provided with the predictions (explainability).\n", + "- `state`: The state of the prediction job (pending, running, etc).\n", + "\n", + "Since this call will take a few moments to execute, you will likely get `JobState.JOB_STATE_PENDING` for `state`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_request:custom" + }, + "outputs": [], + "source": [ + "BATCH_MODEL = \"boston_batch-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_batch_prediction_job(\n", + " display_name,\n", + " model_name,\n", + " gcs_source_uri,\n", + " gcs_destination_output_uri_prefix,\n", + " parameters=None,\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " batch_prediction_job = {\n", + " \"display_name\": display_name,\n", + " # Format: 'projects/{project}/locations/{location}/models/{model_id}'\n", + " \"model\": model_name,\n", + " \"model_parameters\": json_format.ParseDict(parameters, Value()),\n", + " \"input_config\": {\n", + " \"instances_format\": IN_FORMAT,\n", + " \"gcs_source\": {\"uris\": [gcs_source_uri]},\n", + " },\n", + " \"output_config\": {\n", + " \"predictions_format\": OUT_FORMAT,\n", + " \"gcs_destination\": {\"output_uri_prefix\": gcs_destination_output_uri_prefix},\n", + " },\n", + " \"dedicated_resources\": {\n", + " \"machine_spec\": machine_spec,\n", + " \"starting_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " },\n", + " \"generate_explanation\": True,\n", + " }\n", + " response = clients[\"job\"].create_batch_prediction_job(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + " )\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try:\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", response.labels)\n", + " return response\n", + "\n", + "\n", + "IN_FORMAT = \"jsonl\"\n", + "OUT_FORMAT = \"jsonl\"\n", + "\n", + "response = create_batch_prediction_job(\n", + " BATCH_MODEL, model_to_deploy_id, gcs_input_uri, BUCKET_NAME\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batch_job_id:response" + }, + "source": [ + "Now get the unique identifier for the batch prediction job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batch_job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the batch job\n", + "batch_job_id = response.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split(\"/\")[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_batch_prediction_job" + }, + "source": [ + "### Get information on a batch prediction job\n", + "\n", + "Use this helper function `get_batch_prediction_job`, with the following paramter:\n", + "\n", + "- `job_name`: The Vertex fully qualified identifier for the batch prediction job.\n", + "\n", + "The helper function calls the job client service's `get_batch_prediction_job` method, with the following paramter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the batch prediction job. In this tutorial, you will pass it the Vertex fully qualified identifier for your batch prediction job -- `batch_job_id`\n", + "\n", + "The helper function will return the Cloud Storage path to where the predictions are stored -- `gcs_destination`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_batch_prediction_job" + }, + "outputs": [], + "source": [ + "def get_batch_prediction_job(job_name, silent=False):\n", + " response = clients[\"job\"].get_batch_prediction_job(name=job_name)\n", + " if silent:\n", + " return response.output_config.gcs_destination.output_uri_prefix, response.state\n", + "\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try: # not all data types support explanations\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" error:\", response.error)\n", + " gcs_destination = response.output_config.gcs_destination\n", + " print(\" gcs_destination\")\n", + " print(\" output_uri_prefix:\", gcs_destination.output_uri_prefix)\n", + " return gcs_destination.output_uri_prefix, response.state\n", + "\n", + "\n", + "predictions, state = get_batch_prediction_job(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_the_explanations:custom,tabular" + }, + "source": [ + "### Get the predictions\n", + "\n", + "When the batch prediction is done processing, the job state will be `JOB_STATE_SUCCEEDED`.\n", + "\n", + "Finally you view the predictions stored at the Cloud Storage path you set as output. The predictions will be in a JSONL format, which you indicated at the time you made the batch prediction job, under a subfolder starting with the name `prediction`, and under that folder will be a file called `prediction.results-xxxxx-of-xxxxx`.\n", + "\n", + "Now display (cat) the contents. You will see multiple JSON objects, one for each prediction.\n", + "\n", + "Finally you view the explanations stored at the Cloud Storage path you set as output. The explanations will be in a JSONL format, which you indicated at the time you made the batch explanation job, under a subfolder starting with the name `prediction`, and under that folder will be a file called `explanation-results-xxxx-of-xxxx`.\n", + "\n", + "Let's display (cat) the contents. You will a row for each prediction -- in this case, there is just one row. The row contains:\n", + "\n", + "- `dense_input`: The input for the prediction.\n", + "- `prediction`: The predicted value." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_the_explanations:custom,tabular" + }, + "outputs": [], + "source": [ + "def get_latest_predictions(gcs_out_dir):\n", + " \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n", + " folders = !gsutil ls $gcs_out_dir\n", + " latest = \"\"\n", + " for folder in folders:\n", + " subfolder = folder.split(\"/\")[-2]\n", + " if subfolder.startswith(\"prediction-\"):\n", + " if subfolder > latest:\n", + " latest = folder[:-1]\n", + " return latest\n", + "\n", + "\n", + "while True:\n", + " predictions, state = get_batch_prediction_job(batch_job_id, True)\n", + " if state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", state)\n", + " if state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " folder = get_latest_predictions(predictions)\n", + " ! gsutil ls $folder/explanation.results*\n", + "\n", + " print(\"Results:\")\n", + " ! gsutil cat $folder/explanation.results*\n", + "\n", + " print(\"Errors:\")\n", + " ! gsutil cat $folder/prediction.errors*\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_custom_tabular_regression_batch_explain.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_custom_tabular_regression_online.ipynb b/notebooks/community/gapic/custom/showcase_custom_tabular_regression_online.ipynb new file mode 100644 index 000000000..d84a33004 --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_custom_tabular_regression_online.ipynb @@ -0,0 +1,2121 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Custom training tabular regression model for online prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to train and deploy a custom tabular regression model for online prediction." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:custom,boston,lrg" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Boston Housing Prices dataset](https://www.cs.toronto.edu/~delve/data/boston/bostonDetail.html). The version of the dataset you will use in this tutorial is built into TensorFlow. The trained model predicts the median price of a house in units of 1K USD." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,training,online_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create a custom model from a Python script in a Google prebuilt Docker container using the Vertex client library, and then do a prediction on the deployed model by sending data. You can alternatively create custom models using `gcloud` command-line tool or online using Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex custom job for training a model.\n", + "- Train a TensorFlow model.\n", + "- Retrieve and load the model artifacts.\n", + "- View the model evaluation.\n", + "- Upload the model as a Vertex `Model` resource.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for training and prediction.\n", + "\n", + "Set the variables `TRAIN_GPU/TRAIN_NGPU` and `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU.\n", + "\n", + "*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3 -- which is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n", + " TRAIN_GPU, TRAIN_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n", + " )\n", + "else:\n", + " TRAIN_GPU, TRAIN_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (None, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:training,prediction" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for training and prediction\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-3:latest`\n", + " - TensorFlow 2.4\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-4:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/training/xgboost-cpu.1-1`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/training/scikit-learn-cpu.0-23:latest`\n", + " - Pytorch\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-5:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-6:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-7:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for training](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers).\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-3:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-90:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-82:latest`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-23:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-22:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-20:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for prediction](https://cloud.google.com/vertex-ai/docs/predictions/pre-built-containers)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf2-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf2-cpu.{}\".format(TF)\n", + "else:\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "TRAIN_IMAGE = \"gcr.io/cloud-aiplatform/training/{}:latest\".format(TRAIN_VERSION)\n", + "DEPLOY_IMAGE = \"gcr.io/cloud-aiplatform/prediction/{}:latest\".format(DEPLOY_VERSION)\n", + "\n", + "print(\"Training:\", TRAIN_IMAGE, TRAIN_GPU, TRAIN_NGPU)\n", + "print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU, DEPLOY_NGPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:training,prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for training and prediction.\n", + "\n", + "- Set the variables `TRAIN_COMPUTE` and `DEPLOY_COMPUTE` to configure the compute resources for the VMs you will use for for training and prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: The following is not supported for training:*\n", + "\n", + " - `standard`: 2 vCPUs\n", + " - `highcpu`: 2, 4 and 8 vCPUs\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_TRAIN_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Train machine type\", TRAIN_COMPUTE)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own custom model and training for Boston Housing." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:custom" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Model Service for `Model` resources.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:custom" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"job\"] = create_job_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_model" + }, + "source": [ + "## Train a model\n", + "\n", + "There are two ways you can train a custom model using a container image:\n", + "\n", + "- **Use a Google Cloud prebuilt container**. If you use a prebuilt container, you will additionally specify a Python package to install into the container image. This Python package contains your code for training a custom model.\n", + "\n", + "- **Use your own custom container image**. If you use your own container, the container needs to contain your code for training a custom model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_specification:prebuilt_container" + }, + "source": [ + "## Prepare your custom job specification\n", + "\n", + "Now that your clients are ready, your first step is to create a Job Specification for your custom training job. The job specification will consist of the following:\n", + "\n", + "- `worker_pool_spec` : The specification of the type of machine(s) you will use for training and how many (single or distributed)\n", + "- `python_package_spec` : The specification of the Python package to be installed with the pre-built container." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "source": [ + "### Prepare your machine specification\n", + "\n", + "Now define the machine specification for your custom training job. This tells Vertex what type of machine instance to provision for the training.\n", + " - `machine_type`: The type of GCP instance to provision -- e.g., n1-standard-8.\n", + " - `accelerator_type`: The type, if any, of hardware accelerator. In this tutorial if you previously set the variable `TRAIN_GPU != None`, you are using a GPU; otherwise you will use a CPU.\n", + " - `accelerator_count`: The number of accelerators." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "outputs": [], + "source": [ + "if TRAIN_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": TRAIN_COMPUTE,\n", + " \"accelerator_type\": TRAIN_GPU,\n", + " \"accelerator_count\": TRAIN_NGPU,\n", + " }\n", + "else:\n", + " machine_spec = {\"machine_type\": TRAIN_COMPUTE, \"accelerator_count\": 0}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "source": [ + "### Prepare your disk specification\n", + "\n", + "(optional) Now define the disk specification for your custom training job. This tells Vertex what type and size of disk to provision in each machine instance for the training.\n", + "\n", + " - `boot_disk_type`: Either SSD or Standard. SSD is faster, and Standard is less expensive. Defaults to SSD.\n", + " - `boot_disk_size_gb`: Size of disk in GB." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "outputs": [], + "source": [ + "DISK_TYPE = \"pd-ssd\" # [ pd-ssd, pd-standard]\n", + "DISK_SIZE = 200 # GB\n", + "\n", + "disk_spec = {\"boot_disk_type\": DISK_TYPE, \"boot_disk_size_gb\": DISK_SIZE}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container,tabular" + }, + "source": [ + "### Define the worker pool specification\n", + "\n", + "Next, you define the worker pool specification for your custom training job. The worker pool specification will consist of the following:\n", + "\n", + "- `replica_count`: The number of instances to provision of this machine type.\n", + "- `machine_spec`: The hardware specification.\n", + "- `disk_spec` : (optional) The disk storage specification.\n", + "\n", + "- `python_package`: The Python training package to install on the VM instance(s) and which Python module to invoke, along with command line arguments for the Python module.\n", + "\n", + "Let's dive deeper now into the python package specification:\n", + "\n", + "-`executor_image_spec`: This is the docker image which is configured for your custom training job.\n", + "\n", + "-`package_uris`: This is a list of the locations (URIs) of your python training packages to install on the provisioned instance. The locations need to be in a Cloud Storage bucket. These can be either individual python files or a zip (archive) of an entire package. In the later case, the job service will unzip (unarchive) the contents into the docker image.\n", + "\n", + "-`python_module`: The Python module (script) to invoke for running the custom training job. In this example, you will be invoking `trainer.task.py` -- note that it was not neccessary to append the `.py` suffix.\n", + "\n", + "-`args`: The command line arguments to pass to the corresponding Pythom module. In this example, you will be setting:\n", + " - `\"--model-dir=\" + MODEL_DIR` : The Cloud Storage location where to store the model artifacts. There are two ways to tell the training script where to save the model artifacts:\n", + " - direct: You pass the Cloud Storage location as a command line argument to your training script (set variable `DIRECT = True`), or\n", + " - indirect: The service passes the Cloud Storage location as the environment variable `AIP_MODEL_DIR` to your training script (set variable `DIRECT = False`). In this case, you tell the service the model artifact location in the job specification.\n", + " - `\"--epochs=\" + EPOCHS`: The number of epochs for training.\n", + " - `\"--steps=\" + STEPS`: The number of steps (batches) per epoch.\n", + " - `\"--distribute=\" + TRAIN_STRATEGY\"` : The training distribution strategy to use for single or distributed training.\n", + " - `\"single\"`: single device.\n", + " - `\"mirror\"`: all GPU devices on a single compute instance.\n", + " - `\"multi\"`: all GPU devices on all compute instances.\n", + " - `\"--param-file=\" + PARAM_FILE`: The Cloud Storage location for storing feature normalization values." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container,tabular" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_job_\" + TIMESTAMP\n", + "MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, JOB_NAME)\n", + "\n", + "if not TRAIN_NGPU or TRAIN_NGPU < 2:\n", + " TRAIN_STRATEGY = \"single\"\n", + "else:\n", + " TRAIN_STRATEGY = \"mirror\"\n", + "\n", + "EPOCHS = 20\n", + "STEPS = 100\n", + "\n", + "PARAM_FILE = BUCKET_NAME + \"/params.txt\"\n", + "\n", + "DIRECT = True\n", + "if DIRECT:\n", + " CMDARGS = [\n", + " \"--model-dir=\" + MODEL_DIR,\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " \"--param-file=\" + PARAM_FILE,\n", + " ]\n", + "else:\n", + " CMDARGS = [\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " \"--param-file=\" + PARAM_FILE,\n", + " ]\n", + "\n", + "worker_pool_spec = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": machine_spec,\n", + " \"disk_spec\": disk_spec,\n", + " \"python_package_spec\": {\n", + " \"executor_image_uri\": TRAIN_IMAGE,\n", + " \"package_uris\": [BUCKET_NAME + \"/trainer_boston.tar.gz\"],\n", + " \"python_module\": \"trainer.task\",\n", + " \"args\": CMDARGS,\n", + " },\n", + " }\n", + "]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "assemble_custom_job_specification" + }, + "source": [ + "### Assemble a job specification\n", + "\n", + "Now assemble the complete description for the custom job specification:\n", + "\n", + "- `display_name`: The human readable name you assign to this custom job.\n", + "- `job_spec`: The specification for the custom job.\n", + " - `worker_pool_specs`: The specification for the machine VM instances.\n", + " - `base_output_directory`: This tells the service the Cloud Storage location where to save the model artifacts (when variable `DIRECT = False`). The service will then pass the location to the training script as the environment variable `AIP_MODEL_DIR`, and the path will be of the form:\n", + "\n", + " /model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "assemble_custom_job_specification" + }, + "outputs": [], + "source": [ + "if DIRECT:\n", + " job_spec = {\"worker_pool_specs\": worker_pool_spec}\n", + "else:\n", + " job_spec = {\n", + " \"worker_pool_specs\": worker_pool_spec,\n", + " \"base_output_directory\": {\"output_uri_prefix\": MODEL_DIR},\n", + " }\n", + "\n", + "custom_job = {\"display_name\": JOB_NAME, \"job_spec\": job_spec}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "examine_training_package" + }, + "source": [ + "### Examine the training package\n", + "\n", + "#### Package layout\n", + "\n", + "Before you start the training, you will look at how a Python package is assembled for a custom training job. When unarchived, the package contains the following directory/file layout.\n", + "\n", + "- PKG-INFO\n", + "- README.md\n", + "- setup.cfg\n", + "- setup.py\n", + "- trainer\n", + " - \\_\\_init\\_\\_.py\n", + " - task.py\n", + "\n", + "The files `setup.cfg` and `setup.py` are the instructions for installing the package into the operating environment of the Docker image.\n", + "\n", + "The file `trainer/task.py` is the Python script for executing the custom training job. *Note*, when we referred to it in the worker pool specification, we replace the directory slash with a dot (`trainer.task`) and dropped the file suffix (`.py`).\n", + "\n", + "#### Package Assembly\n", + "\n", + "In the following cells, you will assemble the training package." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "examine_training_package" + }, + "outputs": [], + "source": [ + "# Make folder for Python training script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\ntag_build =\\n\\ntag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'tensorflow_datasets==1.3.0',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\nName: Boston Housing tabular regression\\n\\nVersion: 0.0.0\\n\\nSummary: Demostration training script\\n\\nHome-page: www.google.com\\n\\nAuthor: Google\\n\\nAuthor-email: aferlitsch@google.com\\n\\nLicense: Public\\n\\nDescription: Demo\\n\\nPlatform: Vertex\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:boston" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the training script task.py. I won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n", + "- Loads Boston Housing dataset from TF.Keras builtin datasets\n", + "- Builds a simple deep neural network model using TF.Keras model API.\n", + "- Compiles the model (`compile()`).\n", + "- Sets a training distribution strategy according to the argument `args.distribute`.\n", + "- Trains the model (`fit()`) with epochs specified by `args.epochs`.\n", + "- Saves the trained model (`save(args.model_dir)`) to the specified model directory.\n", + "- Saves the maximum value for each feature `f.write(str(params))` to the specified parameters file." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:boston" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Single, Mirror and Multi-Machine Distributed Training for Boston Housing\n", + "\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "import numpy as np\n", + "import argparse\n", + "import os\n", + "import sys\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv('AIP_MODEL_DIR'), type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.001, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=20, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=100, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "parser.add_argument('--param-file', dest='param_file',\n", + " default='/tmp/param.txt', type=str,\n", + " help='Output file for parameters')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "\n", + "# Single Machine, single compute device\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "# Single Machine, multiple compute device\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "# Multiple Machine, multiple compute device\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "# Multi-worker configuration\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "\n", + "def make_dataset():\n", + "\n", + " # Scaling Boston Housing data features\n", + " def scale(feature):\n", + " max = np.max(feature)\n", + " feature = (feature / max).astype(np.float)\n", + " return feature, max\n", + "\n", + " (x_train, y_train), (x_test, y_test) = tf.keras.datasets.boston_housing.load_data(\n", + " path=\"boston_housing.npz\", test_split=0.2, seed=113\n", + " )\n", + " params = []\n", + " for _ in range(13):\n", + " x_train[_], max = scale(x_train[_])\n", + " x_test[_], _ = scale(x_test[_])\n", + " params.append(max)\n", + "\n", + " # store the normalization (max) value for each feature\n", + " with tf.io.gfile.GFile(args.param_file, 'w') as f:\n", + " f.write(str(params))\n", + " return (x_train, y_train), (x_test, y_test)\n", + "\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_dnn_model():\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Dense(128, activation='relu', input_shape=(13,)),\n", + " tf.keras.layers.Dense(128, activation='relu'),\n", + " tf.keras.layers.Dense(1, activation='linear')\n", + " ])\n", + " model.compile(\n", + " loss='mse',\n", + " optimizer=tf.keras.optimizers.RMSprop(learning_rate=args.lr))\n", + " return model\n", + "\n", + "NUM_WORKERS = strategy.num_replicas_in_sync\n", + "# Here the batch size scales up by number of workers since\n", + "# `tf.data.Dataset.batch` expects the global batch size.\n", + "BATCH_SIZE = 16\n", + "GLOBAL_BATCH_SIZE = BATCH_SIZE * NUM_WORKERS\n", + "\n", + "with strategy.scope():\n", + " # Creation of dataset, and model building/compiling need to be within\n", + " # `strategy.scope()`.\n", + " model = build_and_compile_dnn_model()\n", + "\n", + "# Train the model\n", + "(x_train, y_train), (x_test, y_test) = make_dataset()\n", + "model.fit(x_train, y_train, epochs=args.epochs, batch_size=GLOBAL_BATCH_SIZE)\n", + "model.save(args.model_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tarball_training_script" + }, + "source": [ + "#### Store training script on your Cloud Storage bucket\n", + "\n", + "Next, you package the training folder into a compressed tar ball, and then store it in your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tarball_training_script" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_boston.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job" + }, + "source": [ + "### Train the model\n", + "\n", + "\n", + "Now start the training of your custom training job on Vertex. Use this helper function `create_custom_job`, which takes the following parameter:\n", + "\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "The helper function calls job client service's `create_custom_job` method, with the following parameters:\n", + "\n", + "-`parent`: The Vertex location path to `Dataset`, `Model` and `Endpoint` resources.\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "You will display a handful of the fields returned in `response` object, with the two that are of most interest are:\n", + "\n", + "`response.name`: The Vertex fully qualified identifier assigned to this custom training job. You save this identifier for using in subsequent steps.\n", + "\n", + "`response.state`: The current state of the custom training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job" + }, + "outputs": [], + "source": [ + "def create_custom_job(custom_job):\n", + " response = clients[\"job\"].create_custom_job(parent=PARENT, custom_job=custom_job)\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = create_custom_job(custom_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "job_id:response" + }, + "source": [ + "Now get the unique identifier for the custom job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom job\n", + "job_id = response.name\n", + "# The short numeric ID for the custom job\n", + "job_short_id = job_id.split(\"/\")[-1]\n", + "\n", + "print(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_custom_job" + }, + "source": [ + "### Get information on a custom job\n", + "\n", + "Next, use this helper function `get_custom_job`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "The helper function calls the job client service's`get_custom_job` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "If you recall, you got the Vertex fully qualified identifier for the custom job in the `response.name` field when you called the `create_custom_job` method, and saved the identifier in the variable `job_id`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_custom_job" + }, + "outputs": [], + "source": [ + "def get_custom_job(name, silent=False):\n", + " response = clients[\"job\"].get_custom_job(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = get_custom_job(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete:custom" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, we will need to know the location of the saved model, which the Python script saved in your local Cloud Storage bucket at `MODEL_DIR + '/saved_model.pb'`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete:custom" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_custom_job(job_id, True)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_path_to_deploy = None\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " if not DIRECT:\n", + " MODEL_DIR = MODEL_DIR + \"/model\"\n", + " model_path_to_deploy = MODEL_DIR\n", + " print(\"Training Time:\", response.update_time - response.create_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model_to_deploy:\", model_path_to_deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "load_saved_model" + }, + "source": [ + "## Load the saved model\n", + "\n", + "Your model is stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Now load it from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n", + "\n", + "To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "load_saved_model" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:tabular" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now let's find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load the Boston Housing test (holdout) data from `tf.keras.datasets`, using the method `load_data()`. This will return the dataset as a tuple of two elements. The first element is the training data and the second is the test data. Each element is also a tuple of two elements: the feature data, and the corresponding labels (median value of owner-occupied home).\n", + "\n", + "You don't need the training data, and hence why we loaded it as `(_, _)`.\n", + "\n", + "Before you can run the data through evaluation, you need to preprocess it:\n", + "\n", + "x_test:\n", + "1. Normalize (rescaling) the data in each column by dividing each value by the maximum value of that column. This will replace each single value with a 32-bit floating point number between 0 and 1." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "evaluate_custom_model:tabular,boston" + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "from tensorflow.keras.datasets import boston_housing\n", + "\n", + "(_, _), (x_test, y_test) = boston_housing.load_data(\n", + " path=\"boston_housing.npz\", test_split=0.2, seed=113\n", + ")\n", + "\n", + "\n", + "def scale(feature):\n", + " max = np.max(feature)\n", + " feature = (feature / max).astype(np.float32)\n", + " return feature\n", + "\n", + "\n", + "# Let's save one data item that has not been scaled\n", + "x_test_notscaled = x_test[0:1].copy()\n", + "\n", + "for _ in range(13):\n", + " x_test[_] = scale(x_test[_])\n", + "x_test = x_test.astype(np.float32)\n", + "\n", + "print(x_test.shape, x_test.dtype, y_test.shape)\n", + "print(\"scaled\", x_test[0])\n", + "print(\"unscaled\", x_test_notscaled)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "perform_evaluation_custom" + }, + "source": [ + "### Perform the model evaluation\n", + "\n", + "Now evaluate how well the model in the custom job did." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "perform_evaluation_custom" + }, + "outputs": [], + "source": [ + "model.evaluate(x_test, y_test)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_dataset_ststistics" + }, + "source": [ + "### Get dataset statistics\n", + "\n", + "The training script is designed to return dataset statistics you will need for serving predictions on data items that have not otherwise been preprocessed -- feature normalization, which is also referred to as rescaling. *Note*, that the `x_test` data was already preprocessed, so we don't need to do additional feature normalization if we use that data from `x_test`.\n", + "\n", + "Instead, we set aside a copy of one data item that was not feature normalized. We will use this subsequently when doing a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_dataset_ststistics" + }, + "outputs": [], + "source": [ + "# Get the rescaling values,.\n", + "with tf.io.gfile.GFile(PARAM_FILE, \"r\") as f:\n", + " rescale = f.read()\n", + "\n", + "# Convert string to floating point list\n", + "rescale = rescale.replace(\"[\", \"\").replace(\"]\", \"\")\n", + "rescale = [float(val) for val in rescale.split(\",\")]\n", + "print(rescale)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_signature" + }, + "source": [ + "## Get the serving function signature\n", + "\n", + "You can get the signatures of your model's input and output layers by reloading the model into memory, and querying it for the signatures corresponding to each layer.\n", + "\n", + "When making a prediction request, you need to route the request to the serving function instead of the model, so you need to know the input layer name of the serving function -- which you will use later when you make a prediction request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_signature" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_path_to_deploy)\n", + "\n", + "serving_input = list(\n", + " loaded.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "print(\"Serving function input:\", serving_input)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "upload_the_model" + }, + "source": [ + "### Upload the model\n", + "\n", + "Use this helper function `upload_model` to upload your model, stored in SavedModel format, up to the `Model` service, which will instantiate a Vertex `Model` resource instance for your model. Once you've done that, you can use the `Model` resource instance in the same way as any other Vertex `Model` resource instance, such as deploying to an `Endpoint` resource for serving predictions.\n", + "\n", + "The helper function takes the following parameters:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` service.\n", + "- `image_uri`: The container image for the model deployment.\n", + "- `model_uri`: The Cloud Storage path to our SavedModel artificat. For this tutorial, this is the Cloud Storage location where the `trainer/task.py` saved the model artifacts, which we specified in the variable `MODEL_DIR`.\n", + "\n", + "The helper function calls the `Model` client service's method `upload_model`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for `Dataset`, `Model` and `Endpoint` resources.\n", + "- `model`: The specification for the Vertex `Model` resource instance.\n", + "\n", + "Let's now dive deeper into the Vertex model specification `model`. This is a dictionary object that consists of the following fields:\n", + "\n", + "- `display_name`: A human readable name for the `Model` resource.\n", + "- `metadata_schema_uri`: Since your model was built without an Vertex `Dataset` resource, you will leave this blank (`''`).\n", + "- `artificat_uri`: The Cloud Storage path where the model is stored in SavedModel format.\n", + "- `container_spec`: This is the specification for the Docker container that will be installed on the `Endpoint` resource, from which the `Model` resource will serve predictions. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + "\n", + "Uploading a model into a Vertex Model resource returns a long running operation, since it may take a few moments. You call response.result(), which is a synchronous call and will return when the Vertex Model resource is ready.\n", + "\n", + "The helper function returns the Vertex fully qualified identifier for the corresponding Vertex Model instance upload_model_response.model. You will save the identifier for subsequent steps in the variable model_to_deploy_id." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "upload_the_model" + }, + "outputs": [], + "source": [ + "IMAGE_URI = DEPLOY_IMAGE\n", + "\n", + "\n", + "def upload_model(display_name, image_uri, model_uri):\n", + " model = {\n", + " \"display_name\": display_name,\n", + " \"metadata_schema_uri\": \"\",\n", + " \"artifact_uri\": model_uri,\n", + " \"container_spec\": {\n", + " \"image_uri\": image_uri,\n", + " \"command\": [],\n", + " \"args\": [],\n", + " \"env\": [{\"name\": \"env_name\", \"value\": \"env_value\"}],\n", + " \"ports\": [{\"container_port\": 8080}],\n", + " \"predict_route\": \"\",\n", + " \"health_route\": \"\",\n", + " },\n", + " }\n", + " response = clients[\"model\"].upload_model(parent=PARENT, model=model)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " upload_model_response = response.result(timeout=180)\n", + " print(\"upload_model_response\")\n", + " print(\" model:\", upload_model_response.model)\n", + " return upload_model_response.model\n", + "\n", + "\n", + "model_to_deploy_id = upload_model(\n", + " \"boston-\" + TIMESTAMP, IMAGE_URI, model_path_to_deploy\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_model" + }, + "source": [ + "### Get `Model` resource information\n", + "\n", + "Now let's get the model information for just your model. Use this helper function `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource.\n", + "\n", + "This helper function calls the Vertex `Model` client service's method `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_model" + }, + "outputs": [], + "source": [ + "def get_model(name):\n", + " response = clients[\"model\"].get_model(name=name)\n", + " print(response)\n", + "\n", + "\n", + "get_model(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:custom" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex custom `Model` resource. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"boston_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:dedicated" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `dedicated_resources`: This refers to how many compute instances (replicas) that are scaled for serving prediction requests.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `min_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:dedicated" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"boston_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " \"machine_spec\": machine_spec,\n", + " },\n", + " \"disable_container_logging\": False,\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_item:test" + }, + "source": [ + "### Get test item\n", + "\n", + "You will use an example out of the test (holdout) portion of the dataset as a test item." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:test,tabular" + }, + "outputs": [], + "source": [ + "test_item = x_test[0]\n", + "test_label = y_test[0]\n", + "print(test_item.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "send_prediction_request:tabular" + }, + "source": [ + "### Send the prediction request\n", + "\n", + "Ok, now you have a test data item. Use this helper function `predict_data`, which takes the parameters:\n", + "\n", + "- `data`: The test data item as a numpy 1D array of floating point values.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `parameters_dict`: Additional parameters for serving.\n", + "\n", + "This function uses the prediction client service and calls the `predict` method with the parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `instances`: A list of instances (data items) to predict.\n", + "- `parameters`: Additional parameters for serving.\n", + "\n", + "To pass the test data to the prediction service, you package it for transmission to the serving binary as follows:\n", + "\n", + " 1. Convert the data item from a 1D numpy array to a 1D Python list.\n", + " 2. Convert the prediction request to a serialized Google protobuf (`json_format.ParseDict()`)\n", + "\n", + "\n", + "Each instance in the prediction request is a dictionary entry of the form:\n", + "\n", + " {input_name: content}\n", + "\n", + "- `input_name`: the name of the input layer of the underlying model.\n", + "- `content`: The data item as a 1D Python list.\n", + "\n", + "Since the `predict()` service can take multiple data items (instances), you will send your single data item as a list of one data item. As a final step, you package the instances list into Google's protobuf format -- which is what we pass to the `predict()` service.\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding image in the request. You will see in the output for each prediction:\n", + "\n", + "- `predictions` -- the predicated median value of a house in units of 1K USD." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "send_prediction_request:tabular" + }, + "outputs": [], + "source": [ + "def predict_data(data, endpoint, parameters_dict):\n", + " parameters = json_format.ParseDict(parameters_dict, Value())\n", + "\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [{serving_input: data.tolist()}]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", prediction)\n", + "\n", + "\n", + "predict_data(test_item, endpoint_id, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_custom_tabular_regression_online.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_custom_tabular_regression_online_container.ipynb b/notebooks/community/gapic/custom/showcase_custom_tabular_regression_online_container.ipynb new file mode 100644 index 000000000..d15d00ee1 --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_custom_tabular_regression_online_container.ipynb @@ -0,0 +1,2215 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Custom training tabular regression model with custom container for online prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom,training,custom_container" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to train using a custom container and deploy a custom tabular regression model for online prediction." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:custom,boston,lrg" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Boston Housing Prices dataset](https://www.cs.toronto.edu/~delve/data/boston/bostonDetail.html). The version of the dataset you will use in this tutorial is built into TensorFlow. The trained model predicts the median price of a house in units of 1K USD." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,training,custom_container,online_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create a custom model from a Python script in a custom Docker container using the Vertex client library, and then do a prediction on the deployed model by sending data. You can alternatively create custom models using `gcloud` command-line tool or online using Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex custom job for training a model.\n", + "- Train a TensorFlow model using a custom container.\n", + "- Retrieve and load the model artifacts.\n", + "- View the model evaluation.\n", + "- Upload the model as a Vertex `Model` resource.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for training and prediction.\n", + "\n", + "Set the variables `TRAIN_GPU/TRAIN_NGPU` and `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU.\n", + "\n", + "*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3 -- which is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n", + " TRAIN_GPU, TRAIN_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n", + " )\n", + "else:\n", + " TRAIN_GPU, TRAIN_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (None, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:prediction" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for prediction\n", + "\n", + "- Set the variable `TF` to the TensorFlow version of the container image. For example, `2-1` would be version 2.1, and `1-15` would be version 1.15. The following list shows some of the pre-built images available:\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-3:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-90:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-82:latest`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-23:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-22:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-20:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for prediction](https://cloud.google.com/vertex-ai/docs/predictions/pre-built-containers)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf2-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf2-cpu.{}\".format(TF)\n", + "else:\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "DEPLOY_IMAGE = \"gcr.io/cloud-aiplatform/prediction/{}:latest\".format(DEPLOY_VERSION)\n", + "\n", + "print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:training,prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for training and prediction.\n", + "\n", + "- Set the variables `TRAIN_COMPUTE` and `DEPLOY_COMPUTE` to configure the compute resources for the VMs you will use for for training and prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: The following is not supported for training:*\n", + "\n", + " - `standard`: 2 vCPUs\n", + " - `highcpu`: 2, 4 and 8 vCPUs\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_TRAIN_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Train machine type\", TRAIN_COMPUTE)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own custom model and training for Boston Housing." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:custom" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Model Service for `Model` resources.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:custom" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"job\"] = create_job_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_model" + }, + "source": [ + "## Train a model\n", + "\n", + "There are two ways you can train a custom model using a container image:\n", + "\n", + "- **Use a Google Cloud prebuilt container**. If you use a prebuilt container, you will additionally specify a Python package to install into the container image. This Python package contains your code for training a custom model.\n", + "\n", + "- **Use your own custom container image**. If you use your own container, the container needs to contain your code for training a custom model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_docker_container:training" + }, + "source": [ + "### Create a Docker file\n", + "\n", + "In this tutorial, you train a Boston Housing model using your own custom container.\n", + "\n", + "To use your own custom container, you build a Docker file. First, you will create a directory for the container components." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "examine_training_package" + }, + "source": [ + "### Examine the training package\n", + "\n", + "#### Package layout\n", + "\n", + "Before you start the training, you will look at how a Python package is assembled for a custom training job. When unarchived, the package contains the following directory/file layout.\n", + "\n", + "- PKG-INFO\n", + "- README.md\n", + "- setup.cfg\n", + "- setup.py\n", + "- trainer\n", + " - \\_\\_init\\_\\_.py\n", + " - task.py\n", + "\n", + "The files `setup.cfg` and `setup.py` are the instructions for installing the package into the operating environment of the Docker image.\n", + "\n", + "The file `trainer/task.py` is the Python script for executing the custom training job. *Note*, when we referred to it in the worker pool specification, we replace the directory slash with a dot (`trainer.task`) and dropped the file suffix (`.py`).\n", + "\n", + "#### Package Assembly\n", + "\n", + "In the following cells, you will assemble the training package." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "examine_training_package" + }, + "outputs": [], + "source": [ + "# Make folder for Python training script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\ntag_build =\\n\\ntag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'tensorflow_datasets==1.3.0',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\nName: Boston Housing tabular regression\\n\\nVersion: 0.0.0\\n\\nSummary: Demostration training script\\n\\nHome-page: www.google.com\\n\\nAuthor: Google\\n\\nAuthor-email: aferlitsch@google.com\\n\\nLicense: Public\\n\\nDescription: Demo\\n\\nPlatform: Vertex\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:boston" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the training script task.py. I won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n", + "- Loads Boston Housing dataset from TF.Keras builtin datasets\n", + "- Builds a simple deep neural network model using TF.Keras model API.\n", + "- Compiles the model (`compile()`).\n", + "- Sets a training distribution strategy according to the argument `args.distribute`.\n", + "- Trains the model (`fit()`) with epochs specified by `args.epochs`.\n", + "- Saves the trained model (`save(args.model_dir)`) to the specified model directory.\n", + "- Saves the maximum value for each feature `f.write(str(params))` to the specified parameters file." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:boston" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Single, Mirror and Multi-Machine Distributed Training for Boston Housing\n", + "\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "import numpy as np\n", + "import argparse\n", + "import os\n", + "import sys\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv('AIP_MODEL_DIR'), type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.001, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=20, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=100, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "parser.add_argument('--param-file', dest='param_file',\n", + " default='/tmp/param.txt', type=str,\n", + " help='Output file for parameters')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "\n", + "# Single Machine, single compute device\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "# Single Machine, multiple compute device\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "# Multiple Machine, multiple compute device\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "# Multi-worker configuration\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "\n", + "def make_dataset():\n", + "\n", + " # Scaling Boston Housing data features\n", + " def scale(feature):\n", + " max = np.max(feature)\n", + " feature = (feature / max).astype(np.float)\n", + " return feature, max\n", + "\n", + " (x_train, y_train), (x_test, y_test) = tf.keras.datasets.boston_housing.load_data(\n", + " path=\"boston_housing.npz\", test_split=0.2, seed=113\n", + " )\n", + " params = []\n", + " for _ in range(13):\n", + " x_train[_], max = scale(x_train[_])\n", + " x_test[_], _ = scale(x_test[_])\n", + " params.append(max)\n", + "\n", + " # store the normalization (max) value for each feature\n", + " with tf.io.gfile.GFile(args.param_file, 'w') as f:\n", + " f.write(str(params))\n", + " return (x_train, y_train), (x_test, y_test)\n", + "\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_dnn_model():\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Dense(128, activation='relu', input_shape=(13,)),\n", + " tf.keras.layers.Dense(128, activation='relu'),\n", + " tf.keras.layers.Dense(1, activation='linear')\n", + " ])\n", + " model.compile(\n", + " loss='mse',\n", + " optimizer=tf.keras.optimizers.RMSprop(learning_rate=args.lr))\n", + " return model\n", + "\n", + "NUM_WORKERS = strategy.num_replicas_in_sync\n", + "# Here the batch size scales up by number of workers since\n", + "# `tf.data.Dataset.batch` expects the global batch size.\n", + "BATCH_SIZE = 16\n", + "GLOBAL_BATCH_SIZE = BATCH_SIZE * NUM_WORKERS\n", + "\n", + "with strategy.scope():\n", + " # Creation of dataset, and model building/compiling need to be within\n", + " # `strategy.scope()`.\n", + " model = build_and_compile_dnn_model()\n", + "\n", + "# Train the model\n", + "(x_train, y_train), (x_test, y_test) = make_dataset()\n", + "model.fit(x_train, y_train, epochs=args.epochs, batch_size=GLOBAL_BATCH_SIZE)\n", + "model.save(args.model_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "write_docker_file:training,tf-dlvm" + }, + "source": [ + "#### Write the Docker file contents\n", + "\n", + "Your first step in containerizing your code is to create a Docker file. In your Docker you’ll include all the commands needed to run your container image. It’ll install all the libraries you’re using and set up the entry point for your training code.\n", + "\n", + "1. Install a pre-defined container image from TensorFlow repository for deep learning images.\n", + "2. Copies in the Python training code, to be shown subsequently.\n", + "3. Sets the entry into the Python training script as `trainer/task.py`. Note, the `.py` is dropped in the ENTRYPOINT command, as it is implied." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "write_docker_file:training,tf-dlvm" + }, + "outputs": [], + "source": [ + "%%writefile custom/Dockerfile\n", + "\n", + "FROM gcr.io/deeplearning-platform-release/tf2-cpu.2-1\n", + "WORKDIR /root\n", + "\n", + "WORKDIR /\n", + "\n", + "# Copies the trainer code to the docker image.\n", + "COPY trainer /trainer\n", + "\n", + "# Sets up the entry point to invoke the trainer.\n", + "ENTRYPOINT [\"python\", \"-m\", \"trainer.task\"]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "name_container:training" + }, + "source": [ + "#### Build the container locally\n", + "\n", + "Next, you will provide a name for your customer container that you will use when you submit it to the Google Container Registry." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "name_container:training" + }, + "outputs": [], + "source": [ + "TRAIN_IMAGE = \"gcr.io/\" + PROJECT_ID + \"/boston:v1\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "build_container:training" + }, + "source": [ + "Next, build the container." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "build_container:training" + }, + "outputs": [], + "source": [ + "! docker build custom -t $TRAIN_IMAGE" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "test_container:training" + }, + "source": [ + "#### Test the container locally\n", + "\n", + "Run the container within your notebook instance to ensure it’s working correctly. You will run it for 5 epochs." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "test_container:training" + }, + "outputs": [], + "source": [ + "! docker run $TRAIN_IMAGE --epochs=5" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "register_container:training" + }, + "source": [ + "#### Register the custom container\n", + "\n", + "When you’ve finished running the container locally, push it to Google Container Registry." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "register_container:training" + }, + "outputs": [], + "source": [ + "! docker push $TRAIN_IMAGE" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tarball_training_script" + }, + "source": [ + "#### Store training script on your Cloud Storage bucket\n", + "\n", + "Next, you package the training folder into a compressed tar ball, and then store it in your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tarball_training_script" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_boston.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_specification:custom_container" + }, + "source": [ + "## Prepare your custom job specification\n", + "\n", + "Now that your clients are ready, your first step is to create a Job Specification for your custom training job. The job specification will consist of the following:\n", + "\n", + "- `worker_pool_spec` : The specification of the type of machine(s) you will use for training and how many (single or distributed)\n", + "- `container_spec` : The specification of the custom container." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "source": [ + "### Prepare your machine specification\n", + "\n", + "Now define the machine specification for your custom training job. This tells Vertex what type of machine instance to provision for the training.\n", + " - `machine_type`: The type of GCP instance to provision -- e.g., n1-standard-8.\n", + " - `accelerator_type`: The type, if any, of hardware accelerator. In this tutorial if you previously set the variable `TRAIN_GPU != None`, you are using a GPU; otherwise you will use a CPU.\n", + " - `accelerator_count`: The number of accelerators." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "outputs": [], + "source": [ + "if TRAIN_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": TRAIN_COMPUTE,\n", + " \"accelerator_type\": TRAIN_GPU,\n", + " \"accelerator_count\": TRAIN_NGPU,\n", + " }\n", + "else:\n", + " machine_spec = {\"machine_type\": TRAIN_COMPUTE, \"accelerator_count\": 0}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "source": [ + "### Prepare your disk specification\n", + "\n", + "(optional) Now define the disk specification for your custom training job. This tells Vertex what type and size of disk to provision in each machine instance for the training.\n", + "\n", + " - `boot_disk_type`: Either SSD or Standard. SSD is faster, and Standard is less expensive. Defaults to SSD.\n", + " - `boot_disk_size_gb`: Size of disk in GB." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "outputs": [], + "source": [ + "DISK_TYPE = \"pd-ssd\" # [ pd-ssd, pd-standard]\n", + "DISK_SIZE = 200 # GB\n", + "\n", + "disk_spec = {\"boot_disk_type\": DISK_TYPE, \"boot_disk_size_gb\": DISK_SIZE}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "prepare_custom_container_specification" + }, + "source": [ + "### Prepare your container specification\n", + "\n", + "Now define the container specification for your custom training container:\n", + "\n", + "- `image_uri`: The custom container image.\n", + "- `args`: The command-line arguments to pass to the executable that is set as the entry point into the container.\n", + " - `--model-dir` : For our demonstrations, we use this command-line argument to specify where to store the model artifacts.\n", + " - direct: You pass the Cloud Storage location as a command line argument to your training script (set variable `DIRECT = True`), or\n", + " - indirect: The service passes the Cloud Storage location as the environment variable `AIP_MODEL_DIR` to your training script (set variable `DIRECT = False`). In this case, you tell the service the model artifact location in the job specification.\n", + " - `\"--epochs=\" + EPOCHS`: The number of epochs for training.\n", + " - `\"--steps=\" + STEPS`: The number of steps per epoch." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "prepare_custom_container_specification" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"_custom_container\" + TIMESTAMP\n", + "MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, JOB_NAME)\n", + "\n", + "EPOCHS = 20\n", + "STEPS = 100\n", + "\n", + "DIRECT = True\n", + "if DIRECT:\n", + " CMDARGS = [\n", + " \"--model-dir=\" + MODEL_DIR,\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " ]\n", + "else:\n", + " CMDARGS = [\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " ]\n", + "\n", + "container_spec = {\n", + " \"image_uri\": TRAIN_IMAGE,\n", + " \"args\": CMDARGS,\n", + "}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_worker_pool_specification:custom_container" + }, + "source": [ + "### Define the worker pool specification\n", + "\n", + "Next, you define the worker pool specification for your custom training job. The worker pool specification will consist of the following:\n", + "\n", + "- `replica_count`: The number of instances to provision of this machine type.\n", + "- `machine_spec`: The hardware specification.\n", + "- `disk_spec` : (optional) The disk storage specification.\n", + "- `container_spec`: The Docker container to install on the VM instance(s)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_worker_pool_specification:custom_container" + }, + "outputs": [], + "source": [ + "worker_pool_spec = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": machine_spec,\n", + " \"container_spec\": container_spec,\n", + " \"disk_spec\": disk_spec,\n", + " }\n", + "]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "assemble_custom_job_specification" + }, + "source": [ + "### Assemble a job specification\n", + "\n", + "Now assemble the complete description for the custom job specification:\n", + "\n", + "- `display_name`: The human readable name you assign to this custom job.\n", + "- `job_spec`: The specification for the custom job.\n", + " - `worker_pool_specs`: The specification for the machine VM instances.\n", + " - `base_output_directory`: This tells the service the Cloud Storage location where to save the model artifacts (when variable `DIRECT = False`). The service will then pass the location to the training script as the environment variable `AIP_MODEL_DIR`, and the path will be of the form:\n", + "\n", + " /model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "assemble_custom_job_specification" + }, + "outputs": [], + "source": [ + "if DIRECT:\n", + " job_spec = {\"worker_pool_specs\": worker_pool_spec}\n", + "else:\n", + " job_spec = {\n", + " \"worker_pool_specs\": worker_pool_spec,\n", + " \"base_output_directory\": {\"output_uri_prefix\": MODEL_DIR},\n", + " }\n", + "\n", + "custom_job = {\"display_name\": JOB_NAME, \"job_spec\": job_spec}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job" + }, + "source": [ + "### Train the model\n", + "\n", + "\n", + "Now start the training of your custom training job on Vertex. Use this helper function `create_custom_job`, which takes the following parameter:\n", + "\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "The helper function calls job client service's `create_custom_job` method, with the following parameters:\n", + "\n", + "-`parent`: The Vertex location path to `Dataset`, `Model` and `Endpoint` resources.\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "You will display a handful of the fields returned in `response` object, with the two that are of most interest are:\n", + "\n", + "`response.name`: The Vertex fully qualified identifier assigned to this custom training job. You save this identifier for using in subsequent steps.\n", + "\n", + "`response.state`: The current state of the custom training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job" + }, + "outputs": [], + "source": [ + "def create_custom_job(custom_job):\n", + " response = clients[\"job\"].create_custom_job(parent=PARENT, custom_job=custom_job)\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = create_custom_job(custom_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "job_id:response" + }, + "source": [ + "Now get the unique identifier for the custom job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom job\n", + "job_id = response.name\n", + "# The short numeric ID for the custom job\n", + "job_short_id = job_id.split(\"/\")[-1]\n", + "\n", + "print(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_custom_job" + }, + "source": [ + "### Get information on a custom job\n", + "\n", + "Next, use this helper function `get_custom_job`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "The helper function calls the job client service's`get_custom_job` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "If you recall, you got the Vertex fully qualified identifier for the custom job in the `response.name` field when you called the `create_custom_job` method, and saved the identifier in the variable `job_id`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_custom_job" + }, + "outputs": [], + "source": [ + "def get_custom_job(name, silent=False):\n", + " response = clients[\"job\"].get_custom_job(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = get_custom_job(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete:custom" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, we will need to know the location of the saved model, which the Python script saved in your local Cloud Storage bucket at `MODEL_DIR + '/saved_model.pb'`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete:custom" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_custom_job(job_id, True)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_path_to_deploy = None\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " if not DIRECT:\n", + " MODEL_DIR = MODEL_DIR + \"/model\"\n", + " model_path_to_deploy = MODEL_DIR\n", + " print(\"Training Time:\", response.update_time - response.create_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model_to_deploy:\", model_path_to_deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "load_saved_model" + }, + "source": [ + "## Load the saved model\n", + "\n", + "Your model is stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Now load it from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n", + "\n", + "To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "load_saved_model" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:tabular" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now let's find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load the Boston Housing test (holdout) data from `tf.keras.datasets`, using the method `load_data()`. This will return the dataset as a tuple of two elements. The first element is the training data and the second is the test data. Each element is also a tuple of two elements: the feature data, and the corresponding labels (median value of owner-occupied home).\n", + "\n", + "You don't need the training data, and hence why we loaded it as `(_, _)`.\n", + "\n", + "Before you can run the data through evaluation, you need to preprocess it:\n", + "\n", + "x_test:\n", + "1. Normalize (rescaling) the data in each column by dividing each value by the maximum value of that column. This will replace each single value with a 32-bit floating point number between 0 and 1." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "evaluate_custom_model:tabular,boston" + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "from tensorflow.keras.datasets import boston_housing\n", + "\n", + "(_, _), (x_test, y_test) = boston_housing.load_data(\n", + " path=\"boston_housing.npz\", test_split=0.2, seed=113\n", + ")\n", + "\n", + "\n", + "def scale(feature):\n", + " max = np.max(feature)\n", + " feature = (feature / max).astype(np.float32)\n", + " return feature\n", + "\n", + "\n", + "# Let's save one data item that has not been scaled\n", + "x_test_notscaled = x_test[0:1].copy()\n", + "\n", + "for _ in range(13):\n", + " x_test[_] = scale(x_test[_])\n", + "x_test = x_test.astype(np.float32)\n", + "\n", + "print(x_test.shape, x_test.dtype, y_test.shape)\n", + "print(\"scaled\", x_test[0])\n", + "print(\"unscaled\", x_test_notscaled)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "perform_evaluation_custom" + }, + "source": [ + "### Perform the model evaluation\n", + "\n", + "Now evaluate how well the model in the custom job did." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "perform_evaluation_custom" + }, + "outputs": [], + "source": [ + "model.evaluate(x_test, y_test)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "how_serving_function_works" + }, + "source": [ + "## Upload the model for serving\n", + "\n", + "Next, you will upload your TF.Keras model from the custom job to Vertex `Model` service, which will create a Vertex `Model` resource for your custom model. During upload, you need to define a serving function to convert data to the format your model expects. If you send encoded data to Vertex, your serving function ensures that the data is decoded on the model server before it is passed as input to your model.\n", + "\n", + "### How does the serving function work\n", + "\n", + "When you send a request to an online prediction server, the request is received by a HTTP server. The HTTP server extracts the prediction request from the HTTP request content body. The extracted prediction request is forwarded to the serving function. For Google pre-built prediction containers, the request content is passed to the serving function as a `tf.string`.\n", + "\n", + "The serving function consists of two parts:\n", + "\n", + "- `preprocessing function`:\n", + " - Converts the input (`tf.string`) to the input shape and data type of the underlying model (dynamic graph).\n", + " - Performs the same preprocessing of the data that was done during training the underlying model -- e.g., normalizing, scaling, etc.\n", + "- `post-processing function`:\n", + " - Converts the model output to format expected by the receiving application -- e.q., compresses the output.\n", + " - Packages the output for the the receiving application -- e.g., add headings, make JSON object, etc.\n", + "\n", + "Both the preprocessing and post-processing functions are converted to static graphs which are fused to the model. The output from the underlying model is passed to the post-processing function. The post-processing function passes the converted/packaged output back to the HTTP server. The HTTP server returns the output as the HTTP response content.\n", + "\n", + "One consideration you need to consider when building serving functions for TF.Keras models is that they run as static graphs. That means, you cannot use TF graph operations that require a dynamic graph. If you do, you will get an error during the compile of the serving function which will indicate that you are using an EagerTensor which is not supported." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_signature" + }, + "source": [ + "## Get the serving function signature\n", + "\n", + "You can get the signatures of your model's input and output layers by reloading the model into memory, and querying it for the signatures corresponding to each layer.\n", + "\n", + "When making a prediction request, you need to route the request to the serving function instead of the model, so you need to know the input layer name of the serving function -- which you will use later when you make a prediction request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_signature" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_path_to_deploy)\n", + "\n", + "serving_input = list(\n", + " loaded.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "print(\"Serving function input:\", serving_input)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "upload_the_model" + }, + "source": [ + "### Upload the model\n", + "\n", + "Use this helper function `upload_model` to upload your model, stored in SavedModel format, up to the `Model` service, which will instantiate a Vertex `Model` resource instance for your model. Once you've done that, you can use the `Model` resource instance in the same way as any other Vertex `Model` resource instance, such as deploying to an `Endpoint` resource for serving predictions.\n", + "\n", + "The helper function takes the following parameters:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` service.\n", + "- `image_uri`: The container image for the model deployment.\n", + "- `model_uri`: The Cloud Storage path to our SavedModel artificat. For this tutorial, this is the Cloud Storage location where the `trainer/task.py` saved the model artifacts, which we specified in the variable `MODEL_DIR`.\n", + "\n", + "The helper function calls the `Model` client service's method `upload_model`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for `Dataset`, `Model` and `Endpoint` resources.\n", + "- `model`: The specification for the Vertex `Model` resource instance.\n", + "\n", + "Let's now dive deeper into the Vertex model specification `model`. This is a dictionary object that consists of the following fields:\n", + "\n", + "- `display_name`: A human readable name for the `Model` resource.\n", + "- `metadata_schema_uri`: Since your model was built without an Vertex `Dataset` resource, you will leave this blank (`''`).\n", + "- `artificat_uri`: The Cloud Storage path where the model is stored in SavedModel format.\n", + "- `container_spec`: This is the specification for the Docker container that will be installed on the `Endpoint` resource, from which the `Model` resource will serve predictions. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + "\n", + "Uploading a model into a Vertex Model resource returns a long running operation, since it may take a few moments. You call response.result(), which is a synchronous call and will return when the Vertex Model resource is ready.\n", + "\n", + "The helper function returns the Vertex fully qualified identifier for the corresponding Vertex Model instance upload_model_response.model. You will save the identifier for subsequent steps in the variable model_to_deploy_id." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "upload_the_model" + }, + "outputs": [], + "source": [ + "IMAGE_URI = DEPLOY_IMAGE\n", + "\n", + "\n", + "def upload_model(display_name, image_uri, model_uri):\n", + " model = {\n", + " \"display_name\": display_name,\n", + " \"metadata_schema_uri\": \"\",\n", + " \"artifact_uri\": model_uri,\n", + " \"container_spec\": {\n", + " \"image_uri\": image_uri,\n", + " \"command\": [],\n", + " \"args\": [],\n", + " \"env\": [{\"name\": \"env_name\", \"value\": \"env_value\"}],\n", + " \"ports\": [{\"container_port\": 8080}],\n", + " \"predict_route\": \"\",\n", + " \"health_route\": \"\",\n", + " },\n", + " }\n", + " response = clients[\"model\"].upload_model(parent=PARENT, model=model)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " upload_model_response = response.result(timeout=180)\n", + " print(\"upload_model_response\")\n", + " print(\" model:\", upload_model_response.model)\n", + " return upload_model_response.model\n", + "\n", + "\n", + "model_to_deploy_id = upload_model(\n", + " \"boston-\" + TIMESTAMP, IMAGE_URI, model_path_to_deploy\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_model" + }, + "source": [ + "### Get `Model` resource information\n", + "\n", + "Now let's get the model information for just your model. Use this helper function `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource.\n", + "\n", + "This helper function calls the Vertex `Model` client service's method `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_model" + }, + "outputs": [], + "source": [ + "def get_model(name):\n", + " response = clients[\"model\"].get_model(name=name)\n", + " print(response)\n", + "\n", + "\n", + "get_model(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:custom" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex custom `Model` resource. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"boston_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:dedicated" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `dedicated_resources`: This refers to how many compute instances (replicas) that are scaled for serving prediction requests.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `min_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:dedicated" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"boston_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " \"machine_spec\": machine_spec,\n", + " },\n", + " \"disable_container_logging\": False,\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_item:test,tabular" + }, + "source": [ + "### Get test item\n", + "\n", + "You will use an example out of the test (holdout) portion of the dataset as a test item." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:test,tabular" + }, + "outputs": [], + "source": [ + "test_item = x_test[0]\n", + "test_label = y_test[0]\n", + "print(test_item.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "send_prediction_request:tabular" + }, + "source": [ + "### Send the prediction request\n", + "\n", + "Ok, now you have a test data item. Use this helper function `predict_data`, which takes the parameters:\n", + "\n", + "- `data`: The test data item as a numpy 1D array of floating point values.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `parameters_dict`: Additional parameters for serving.\n", + "\n", + "This function uses the prediction client service and calls the `predict` method with the parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `instances`: A list of instances (data items) to predict.\n", + "- `parameters`: Additional parameters for serving.\n", + "\n", + "To pass the test data to the prediction service, you package it for transmission to the serving binary as follows:\n", + "\n", + " 1. Convert the data item from a 1D numpy array to a 1D Python list.\n", + " 2. Convert the prediction request to a serialized Google protobuf (`json_format.ParseDict()`)\n", + "\n", + "\n", + "Each instance in the prediction request is a dictionary entry of the form:\n", + "\n", + " {input_name: content}\n", + "\n", + "- `input_name`: the name of the input layer of the underlying model.\n", + "- `content`: The data item as a 1D Python list.\n", + "\n", + "Since the `predict()` service can take multiple data items (instances), you will send your single data item as a list of one data item. As a final step, you package the instances list into Google's protobuf format -- which is what we pass to the `predict()` service.\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding image in the request. You will see in the output for each prediction:\n", + "\n", + "- `predictions` -- the predicated median value of a house in units of 1K USD." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "send_prediction_request:tabular" + }, + "outputs": [], + "source": [ + "def predict_data(data, endpoint, parameters_dict):\n", + " parameters = json_format.ParseDict(parameters_dict, Value())\n", + "\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [{serving_input: data.tolist()}]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", prediction)\n", + "\n", + "\n", + "predict_data(test_item, endpoint_id, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_custom_tabular_regression_online_container.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_custom_tabular_regression_online_explain.ipynb b/notebooks/community/gapic/custom/showcase_custom_tabular_regression_online_explain.ipynb new file mode 100644 index 000000000..be0c5dac2 --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_custom_tabular_regression_online_explain.ipynb @@ -0,0 +1,2498 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Custom training tabular regression model for online prediction with explanation\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom,xai" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to train and deploy a custom tabular regression model for online prediction with explanation." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:custom,boston,lrg" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Boston Housing Prices dataset](https://www.cs.toronto.edu/~delve/data/boston/bostonDetail.html). The version of the dataset you will use in this tutorial is built into TensorFlow. The trained model predicts the median price of a house in units of 1K USD." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,training,online_prediction,xai" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create a custom model from a Python script in a Google prebuilt Docker container using the Vertex client library, and then do a prediction with explanations on the deployed model by sending data. You can alternatively create custom models using `gcloud` command-line tool or online using Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex custom job for training a model.\n", + "- Train a TensorFlow model.\n", + "- Retrieve and load the model artifacts.\n", + "- View the model evaluation.\n", + "- Set explanation parameters.\n", + "- Upload the model as a Vertex `Model` resource.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction with explanation.\n", + "- Undeploy the `Model` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_other_packages" + }, + "source": [ + "### Install other packages\n", + "\n", + "Install other packages required for this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_tabulate" + }, + "outputs": [], + "source": [ + "! pip3 install -U tabulate $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:v1beta1,protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:v1beta1,protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "import google.cloud.aiplatform_v1beta1 as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for training and prediction.\n", + "\n", + "Set the variables `TRAIN_GPU/TRAIN_NGPU` and `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU.\n", + "\n", + "*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3 -- which is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n", + " TRAIN_GPU, TRAIN_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n", + " )\n", + "else:\n", + " TRAIN_GPU, TRAIN_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (None, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:training,prediction" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for training and prediction\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-3:latest`\n", + " - TensorFlow 2.4\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-4:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/training/xgboost-cpu.1-1`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/training/scikit-learn-cpu.0-23:latest`\n", + " - Pytorch\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-5:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-6:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-7:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for training](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers).\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-3:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-90:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-82:latest`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-23:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-22:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-20:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for prediction](https://cloud.google.com/vertex-ai/docs/predictions/pre-built-containers)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf2-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf2-cpu.{}\".format(TF)\n", + "else:\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "TRAIN_IMAGE = \"gcr.io/cloud-aiplatform/training/{}:latest\".format(TRAIN_VERSION)\n", + "DEPLOY_IMAGE = \"gcr.io/cloud-aiplatform/prediction/{}:latest\".format(DEPLOY_VERSION)\n", + "\n", + "print(\"Training:\", TRAIN_IMAGE, TRAIN_GPU, TRAIN_NGPU)\n", + "print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU, DEPLOY_NGPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:training,prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for training and prediction.\n", + "\n", + "- Set the variables `TRAIN_COMPUTE` and `DEPLOY_COMPUTE` to configure the compute resources for the VMs you will use for for training and prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: The following is not supported for training:*\n", + "\n", + " - `standard`: 2 vCPUs\n", + " - `highcpu`: 2, 4 and 8 vCPUs\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_TRAIN_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Train machine type\", TRAIN_COMPUTE)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own custom model and training for Boston Housing." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:custom" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Model Service for `Model` resources.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:custom" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"job\"] = create_job_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_model" + }, + "source": [ + "## Train a model\n", + "\n", + "There are two ways you can train a custom model using a container image:\n", + "\n", + "- **Use a Google Cloud prebuilt container**. If you use a prebuilt container, you will additionally specify a Python package to install into the container image. This Python package contains your code for training a custom model.\n", + "\n", + "- **Use your own custom container image**. If you use your own container, the container needs to contain your code for training a custom model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_specification:prebuilt_container" + }, + "source": [ + "## Prepare your custom job specification\n", + "\n", + "Now that your clients are ready, your first step is to create a Job Specification for your custom training job. The job specification will consist of the following:\n", + "\n", + "- `worker_pool_spec` : The specification of the type of machine(s) you will use for training and how many (single or distributed)\n", + "- `python_package_spec` : The specification of the Python package to be installed with the pre-built container." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "source": [ + "### Prepare your machine specification\n", + "\n", + "Now define the machine specification for your custom training job. This tells Vertex what type of machine instance to provision for the training.\n", + " - `machine_type`: The type of GCP instance to provision -- e.g., n1-standard-8.\n", + " - `accelerator_type`: The type, if any, of hardware accelerator. In this tutorial if you previously set the variable `TRAIN_GPU != None`, you are using a GPU; otherwise you will use a CPU.\n", + " - `accelerator_count`: The number of accelerators." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "outputs": [], + "source": [ + "if TRAIN_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": TRAIN_COMPUTE,\n", + " \"accelerator_type\": TRAIN_GPU,\n", + " \"accelerator_count\": TRAIN_NGPU,\n", + " }\n", + "else:\n", + " machine_spec = {\"machine_type\": TRAIN_COMPUTE, \"accelerator_count\": 0}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "source": [ + "### Prepare your disk specification\n", + "\n", + "(optional) Now define the disk specification for your custom training job. This tells Vertex what type and size of disk to provision in each machine instance for the training.\n", + "\n", + " - `boot_disk_type`: Either SSD or Standard. SSD is faster, and Standard is less expensive. Defaults to SSD.\n", + " - `boot_disk_size_gb`: Size of disk in GB." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "outputs": [], + "source": [ + "DISK_TYPE = \"pd-ssd\" # [ pd-ssd, pd-standard]\n", + "DISK_SIZE = 200 # GB\n", + "\n", + "disk_spec = {\"boot_disk_type\": DISK_TYPE, \"boot_disk_size_gb\": DISK_SIZE}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container" + }, + "source": [ + "### Define the worker pool specification\n", + "\n", + "Next, you define the worker pool specification for your custom training job. The worker pool specification will consist of the following:\n", + "\n", + "- `replica_count`: The number of instances to provision of this machine type.\n", + "- `machine_spec`: The hardware specification.\n", + "- `disk_spec` : (optional) The disk storage specification.\n", + "\n", + "- `python_package`: The Python training package to install on the VM instance(s) and which Python module to invoke, along with command line arguments for the Python module.\n", + "\n", + "Let's dive deeper now into the python package specification:\n", + "\n", + "-`executor_image_spec`: This is the docker image which is configured for your custom training job.\n", + "\n", + "-`package_uris`: This is a list of the locations (URIs) of your python training packages to install on the provisioned instance. The locations need to be in a Cloud Storage bucket. These can be either individual python files or a zip (archive) of an entire package. In the later case, the job service will unzip (unarchive) the contents into the docker image.\n", + "\n", + "-`python_module`: The Python module (script) to invoke for running the custom training job. In this example, you will be invoking `trainer.task.py` -- note that it was not neccessary to append the `.py` suffix.\n", + "\n", + "-`args`: The command line arguments to pass to the corresponding Pythom module. In this example, you will be setting:\n", + " - `\"--model-dir=\" + MODEL_DIR` : The Cloud Storage location where to store the model artifacts. There are two ways to tell the training script where to save the model artifacts:\n", + " - direct: You pass the Cloud Storage location as a command line argument to your training script (set variable `DIRECT = True`), or\n", + " - indirect: The service passes the Cloud Storage location as the environment variable `AIP_MODEL_DIR` to your training script (set variable `DIRECT = False`). In this case, you tell the service the model artifact location in the job specification.\n", + " - `\"--epochs=\" + EPOCHS`: The number of epochs for training.\n", + " - `\"--steps=\" + STEPS`: The number of steps (batches) per epoch.\n", + " - `\"--distribute=\" + TRAIN_STRATEGY\"` : The training distribution strategy to use for single or distributed training.\n", + " - `\"single\"`: single device.\n", + " - `\"mirror\"`: all GPU devices on a single compute instance.\n", + " - `\"multi\"`: all GPU devices on all compute instances." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_job_\" + TIMESTAMP\n", + "MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, JOB_NAME)\n", + "\n", + "if not TRAIN_NGPU or TRAIN_NGPU < 2:\n", + " TRAIN_STRATEGY = \"single\"\n", + "else:\n", + " TRAIN_STRATEGY = \"mirror\"\n", + "\n", + "EPOCHS = 20\n", + "STEPS = 100\n", + "\n", + "DIRECT = True\n", + "if DIRECT:\n", + " CMDARGS = [\n", + " \"--model-dir=\" + MODEL_DIR,\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "else:\n", + " CMDARGS = [\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "\n", + "worker_pool_spec = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": machine_spec,\n", + " \"disk_spec\": disk_spec,\n", + " \"python_package_spec\": {\n", + " \"executor_image_uri\": TRAIN_IMAGE,\n", + " \"package_uris\": [BUCKET_NAME + \"/trainer_boston.tar.gz\"],\n", + " \"python_module\": \"trainer.task\",\n", + " \"args\": CMDARGS,\n", + " },\n", + " }\n", + "]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "assemble_custom_job_specification" + }, + "source": [ + "### Assemble a job specification\n", + "\n", + "Now assemble the complete description for the custom job specification:\n", + "\n", + "- `display_name`: The human readable name you assign to this custom job.\n", + "- `job_spec`: The specification for the custom job.\n", + " - `worker_pool_specs`: The specification for the machine VM instances.\n", + " - `base_output_directory`: This tells the service the Cloud Storage location where to save the model artifacts (when variable `DIRECT = False`). The service will then pass the location to the training script as the environment variable `AIP_MODEL_DIR`, and the path will be of the form:\n", + "\n", + " /model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "assemble_custom_job_specification" + }, + "outputs": [], + "source": [ + "if DIRECT:\n", + " job_spec = {\"worker_pool_specs\": worker_pool_spec}\n", + "else:\n", + " job_spec = {\n", + " \"worker_pool_specs\": worker_pool_spec,\n", + " \"base_output_directory\": {\"output_uri_prefix\": MODEL_DIR},\n", + " }\n", + "\n", + "custom_job = {\"display_name\": JOB_NAME, \"job_spec\": job_spec}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "examine_training_package" + }, + "source": [ + "### Examine the training package\n", + "\n", + "#### Package layout\n", + "\n", + "Before you start the training, you will look at how a Python package is assembled for a custom training job. When unarchived, the package contains the following directory/file layout.\n", + "\n", + "- PKG-INFO\n", + "- README.md\n", + "- setup.cfg\n", + "- setup.py\n", + "- trainer\n", + " - \\_\\_init\\_\\_.py\n", + " - task.py\n", + "\n", + "The files `setup.cfg` and `setup.py` are the instructions for installing the package into the operating environment of the Docker image.\n", + "\n", + "The file `trainer/task.py` is the Python script for executing the custom training job. *Note*, when we referred to it in the worker pool specification, we replace the directory slash with a dot (`trainer.task`) and dropped the file suffix (`.py`).\n", + "\n", + "#### Package Assembly\n", + "\n", + "In the following cells, you will assemble the training package." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "examine_training_package" + }, + "outputs": [], + "source": [ + "# Make folder for Python training script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\ntag_build =\\n\\ntag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'tensorflow_datasets==1.3.0',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\nName: Boston Housing tabular regression\\n\\nVersion: 0.0.0\\n\\nSummary: Demostration training script\\n\\nHome-page: www.google.com\\n\\nAuthor: Google\\n\\nAuthor-email: aferlitsch@google.com\\n\\nLicense: Public\\n\\nDescription: Demo\\n\\nPlatform: Vertex\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:boston" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the training script task.py. I won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n", + "- Loads Boston Housing dataset from TF.Keras builtin datasets\n", + "- Builds a simple deep neural network model using TF.Keras model API.\n", + "- Compiles the model (`compile()`).\n", + "- Sets a training distribution strategy according to the argument `args.distribute`.\n", + "- Trains the model (`fit()`) with epochs specified by `args.epochs`.\n", + "- Saves the trained model (`save(args.model_dir)`) to the specified model directory.\n", + "- Saves the maximum value for each feature `f.write(str(params))` to the specified parameters file." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:boston" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Single, Mirror and Multi-Machine Distributed Training for Boston Housing\n", + "\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "import numpy as np\n", + "import argparse\n", + "import os\n", + "import sys\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv('AIP_MODEL_DIR'), type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.001, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=20, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=100, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "parser.add_argument('--param-file', dest='param_file',\n", + " default='/tmp/param.txt', type=str,\n", + " help='Output file for parameters')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "\n", + "# Single Machine, single compute device\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "# Single Machine, multiple compute device\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "# Multiple Machine, multiple compute device\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "# Multi-worker configuration\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "\n", + "def make_dataset():\n", + "\n", + " # Scaling Boston Housing data features\n", + " def scale(feature):\n", + " max = np.max(feature)\n", + " feature = (feature / max).astype(np.float)\n", + " return feature, max\n", + "\n", + " (x_train, y_train), (x_test, y_test) = tf.keras.datasets.boston_housing.load_data(\n", + " path=\"boston_housing.npz\", test_split=0.2, seed=113\n", + " )\n", + " params = []\n", + " for _ in range(13):\n", + " x_train[_], max = scale(x_train[_])\n", + " x_test[_], _ = scale(x_test[_])\n", + " params.append(max)\n", + "\n", + " # store the normalization (max) value for each feature\n", + " with tf.io.gfile.GFile(args.param_file, 'w') as f:\n", + " f.write(str(params))\n", + " return (x_train, y_train), (x_test, y_test)\n", + "\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_dnn_model():\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Dense(128, activation='relu', input_shape=(13,)),\n", + " tf.keras.layers.Dense(128, activation='relu'),\n", + " tf.keras.layers.Dense(1, activation='linear')\n", + " ])\n", + " model.compile(\n", + " loss='mse',\n", + " optimizer=tf.keras.optimizers.RMSprop(learning_rate=args.lr))\n", + " return model\n", + "\n", + "NUM_WORKERS = strategy.num_replicas_in_sync\n", + "# Here the batch size scales up by number of workers since\n", + "# `tf.data.Dataset.batch` expects the global batch size.\n", + "BATCH_SIZE = 16\n", + "GLOBAL_BATCH_SIZE = BATCH_SIZE * NUM_WORKERS\n", + "\n", + "with strategy.scope():\n", + " # Creation of dataset, and model building/compiling need to be within\n", + " # `strategy.scope()`.\n", + " model = build_and_compile_dnn_model()\n", + "\n", + "# Train the model\n", + "(x_train, y_train), (x_test, y_test) = make_dataset()\n", + "model.fit(x_train, y_train, epochs=args.epochs, batch_size=GLOBAL_BATCH_SIZE)\n", + "model.save(args.model_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tarball_training_script" + }, + "source": [ + "#### Store training script on your Cloud Storage bucket\n", + "\n", + "Next, you package the training folder into a compressed tar ball, and then store it in your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tarball_training_script" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_boston.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job" + }, + "source": [ + "### Train the model\n", + "\n", + "\n", + "Now start the training of your custom training job on Vertex. Use this helper function `create_custom_job`, which takes the following parameter:\n", + "\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "The helper function calls job client service's `create_custom_job` method, with the following parameters:\n", + "\n", + "-`parent`: The Vertex location path to `Dataset`, `Model` and `Endpoint` resources.\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "You will display a handful of the fields returned in `response` object, with the two that are of most interest are:\n", + "\n", + "`response.name`: The Vertex fully qualified identifier assigned to this custom training job. You save this identifier for using in subsequent steps.\n", + "\n", + "`response.state`: The current state of the custom training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job" + }, + "outputs": [], + "source": [ + "def create_custom_job(custom_job):\n", + " response = clients[\"job\"].create_custom_job(parent=PARENT, custom_job=custom_job)\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = create_custom_job(custom_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "job_id:response" + }, + "source": [ + "Now get the unique identifier for the custom job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom job\n", + "job_id = response.name\n", + "# The short numeric ID for the custom job\n", + "job_short_id = job_id.split(\"/\")[-1]\n", + "\n", + "print(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_custom_job" + }, + "source": [ + "### Get information on a custom job\n", + "\n", + "Next, use this helper function `get_custom_job`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "The helper function calls the job client service's`get_custom_job` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "If you recall, you got the Vertex fully qualified identifier for the custom job in the `response.name` field when you called the `create_custom_job` method, and saved the identifier in the variable `job_id`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_custom_job" + }, + "outputs": [], + "source": [ + "def get_custom_job(name, silent=False):\n", + " response = clients[\"job\"].get_custom_job(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = get_custom_job(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete:custom" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, we will need to know the location of the saved model, which the Python script saved in your local Cloud Storage bucket at `MODEL_DIR + '/saved_model.pb'`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete:custom" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_custom_job(job_id, True)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_path_to_deploy = None\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " if not DIRECT:\n", + " MODEL_DIR = MODEL_DIR + \"/model\"\n", + " model_path_to_deploy = MODEL_DIR\n", + " print(\"Training Time:\", response.update_time - response.create_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model_to_deploy:\", model_path_to_deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "load_saved_model" + }, + "source": [ + "## Load the saved model\n", + "\n", + "Your model is stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Now load it from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n", + "\n", + "To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "load_saved_model" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:tabular" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now let's find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load the Boston Housing test (holdout) data from `tf.keras.datasets`, using the method `load_data()`. This will return the dataset as a tuple of two elements. The first element is the training data and the second is the test data. Each element is also a tuple of two elements: the feature data, and the corresponding labels (median value of owner-occupied home).\n", + "\n", + "You don't need the training data, and hence why we loaded it as `(_, _)`.\n", + "\n", + "Before you can run the data through evaluation, you need to preprocess it:\n", + "\n", + "x_test:\n", + "1. Normalize (rescaling) the data in each column by dividing each value by the maximum value of that column. This will replace each single value with a 32-bit floating point number between 0 and 1." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "evaluate_custom_model:tabular,boston" + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "from tensorflow.keras.datasets import boston_housing\n", + "\n", + "(_, _), (x_test, y_test) = boston_housing.load_data(\n", + " path=\"boston_housing.npz\", test_split=0.2, seed=113\n", + ")\n", + "\n", + "\n", + "def scale(feature):\n", + " max = np.max(feature)\n", + " feature = (feature / max).astype(np.float32)\n", + " return feature\n", + "\n", + "\n", + "# Let's save one data item that has not been scaled\n", + "x_test_notscaled = x_test[0:1].copy()\n", + "\n", + "for _ in range(13):\n", + " x_test[_] = scale(x_test[_])\n", + "x_test = x_test.astype(np.float32)\n", + "\n", + "print(x_test.shape, x_test.dtype, y_test.shape)\n", + "print(\"scaled\", x_test[0])\n", + "print(\"unscaled\", x_test_notscaled)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "perform_evaluation_custom" + }, + "source": [ + "### Perform the model evaluation\n", + "\n", + "Now evaluate how well the model in the custom job did." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "perform_evaluation_custom" + }, + "outputs": [], + "source": [ + "model.evaluate(x_test, y_test)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "how_serving_function_works" + }, + "source": [ + "## Upload the model for serving\n", + "\n", + "Next, you will upload your TF.Keras model from the custom job to Vertex `Model` service, which will create a Vertex `Model` resource for your custom model. During upload, you need to define a serving function to convert data to the format your model expects. If you send encoded data to Vertex, your serving function ensures that the data is decoded on the model server before it is passed as input to your model.\n", + "\n", + "### How does the serving function work\n", + "\n", + "When you send a request to an online prediction server, the request is received by a HTTP server. The HTTP server extracts the prediction request from the HTTP request content body. The extracted prediction request is forwarded to the serving function. For Google pre-built prediction containers, the request content is passed to the serving function as a `tf.string`.\n", + "\n", + "The serving function consists of two parts:\n", + "\n", + "- `preprocessing function`:\n", + " - Converts the input (`tf.string`) to the input shape and data type of the underlying model (dynamic graph).\n", + " - Performs the same preprocessing of the data that was done during training the underlying model -- e.g., normalizing, scaling, etc.\n", + "- `post-processing function`:\n", + " - Converts the model output to format expected by the receiving application -- e.q., compresses the output.\n", + " - Packages the output for the the receiving application -- e.g., add headings, make JSON object, etc.\n", + "\n", + "Both the preprocessing and post-processing functions are converted to static graphs which are fused to the model. The output from the underlying model is passed to the post-processing function. The post-processing function passes the converted/packaged output back to the HTTP server. The HTTP server returns the output as the HTTP response content.\n", + "\n", + "One consideration you need to consider when building serving functions for TF.Keras models is that they run as static graphs. That means, you cannot use TF graph operations that require a dynamic graph. If you do, you will get an error during the compile of the serving function which will indicate that you are using an EagerTensor which is not supported." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_signature:xai" + }, + "source": [ + "## Get the serving function signature\n", + "\n", + "You can get the signatures of your model's input and output layers by reloading the model into memory, and querying it for the signatures corresponding to each layer.\n", + "\n", + "When making a prediction request, you need to route the request to the serving function instead of the model, so you need to know the input layer name of the serving function -- which you will use later when you make a prediction request.\n", + "\n", + "You also need to know the name of the serving function's input and output layer for constructing the explanation metadata -- which is discussed subsequently." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_signature:xai" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_path_to_deploy)\n", + "\n", + "serving_input = list(\n", + " loaded.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "print(\"Serving function input:\", serving_input)\n", + "serving_output = list(loaded.signatures[\"serving_default\"].structured_outputs.keys())[0]\n", + "print(\"Serving function output:\", serving_output)\n", + "\n", + "input_name = model.input.name\n", + "print(\"Model input name:\", input_name)\n", + "output_name = model.output.name\n", + "print(\"Model output name:\", output_name)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "explanation_spec" + }, + "source": [ + "### Explanation Specification\n", + "\n", + "To get explanations when doing a prediction, you must enable the explanation capability and set corresponding settings when you upload your custom model to an Vertex `Model` resource. These settings are referred to as the explanation metadata, which consists of:\n", + "\n", + "- `parameters`: This is the specification for the explainability algorithm to use for explanations on your model. You can choose between:\n", + " - Shapley - *Note*, not recommended for image data -- can be very long running\n", + " - XRAI\n", + " - Integrated Gradients\n", + "- `metadata`: This is the specification for how the algoithm is applied on your custom model.\n", + "\n", + "#### Explanation Parameters\n", + "\n", + "Let's first dive deeper into the settings for the explainability algorithm.\n", + "\n", + "#### Shapley\n", + "\n", + "Assigns credit for the outcome to each feature, and considers different permutations of the features. This method provides a sampling approximation of exact Shapley values.\n", + "\n", + "Use Cases:\n", + " - Classification and regression on tabular data.\n", + "\n", + "Parameters:\n", + "\n", + "- `path_count`: This is the number of paths over the features that will be processed by the algorithm. An exact approximation of the Shapley values requires M! paths, where M is the number of features. For the CIFAR10 dataset, this would be 784 (28*28).\n", + "\n", + "For any non-trival number of features, this is too compute expensive. You can reduce the number of paths over the features to M * `path_count`.\n", + "\n", + "#### Integrated Gradients\n", + "\n", + "A gradients-based method to efficiently compute feature attributions with the same axiomatic properties as the Shapley value.\n", + "\n", + "Use Cases:\n", + " - Classification and regression on tabular data.\n", + " - Classification on image data.\n", + "\n", + "Parameters:\n", + "\n", + "- `step_count`: This is the number of steps to approximate the remaining sum. The more steps, the more accurate the integral approximation. The general rule of thumb is 50 steps, but as you increase so does the compute time.\n", + "\n", + "#### XRAI\n", + "\n", + "Based on the integrated gradients method, XRAI assesses overlapping regions of the image to create a saliency map, which highlights relevant regions of the image rather than pixels.\n", + "\n", + "Use Cases:\n", + "\n", + " - Classification on image data.\n", + "\n", + "Parameters:\n", + "\n", + "- `step_count`: This is the number of steps to approximate the remaining sum. The more steps, the more accurate the integral approximation. The general rule of thumb is 50 steps, but as you increase so does the compute time.\n", + "\n", + "In the next code cell, set the variable `XAI` to which explainabilty algorithm you will use on your custom model." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "explanation_spec" + }, + "outputs": [], + "source": [ + "XAI = \"ig\" # [ shapley, ig, xrai ]\n", + "\n", + "if XAI == \"shapley\":\n", + " PARAMETERS = {\"sampled_shapley_attribution\": {\"path_count\": 10}}\n", + "elif XAI == \"ig\":\n", + " PARAMETERS = {\"integrated_gradients_attribution\": {\"step_count\": 50}}\n", + "elif XAI == \"xrai\":\n", + " PARAMETERS = {\"xrai_attribution\": {\"step_count\": 50}}\n", + "\n", + "parameters = aip.ExplanationParameters(PARAMETERS)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "explanation_metadata:tabular" + }, + "source": [ + "#### Explanation Metadata\n", + "\n", + "Let's first dive deeper into the explanation metadata, which consists of:\n", + "\n", + "- `outputs`: A scalar value in the output to attribute -- what to explain. For example, in a probability output \\[0.1, 0.2, 0.7\\] for classification, one wants an explanation for 0.7. Consider the following formulae, where the output is `y` and that is what we want to explain.\n", + "\n", + " y = f(x)\n", + "\n", + "Consider the following formulae, where the outputs are `y` and `z`. Since we can only do attribution for one scalar value, we have to pick whether we want to explain the output `y` or `z`. Assume in this example the model is object detection and y and z are the bounding box and the object classification. You would want to pick which of the two outputs to explain.\n", + "\n", + " y, z = f(x)\n", + "\n", + "The dictionary format for `outputs` is:\n", + "\n", + " { \"outputs\": { \"[your_display_name]\":\n", + " \"output_tensor_name\": [layer]\n", + " }\n", + " }\n", + "\n", + "
\n", + " - [your_display_name]: A human readable name you assign to the output to explain. A common example is \"probability\".
\n", + " - \"output_tensor_name\": The key/value field to identify the output layer to explain.
\n", + " - [layer]: The output layer to explain. In a single task model, like a tabular regressor, it is the last (topmost) layer in the model.\n", + "
\n", + "\n", + "- `inputs`: The features for attribution -- how they contributed to the output. Consider the following formulae, where `a` and `b` are the features. We have to pick which features to explain how the contributed. Assume that this model is deployed for A/B testing, where `a` are the data_items for the prediction and `b` identifies whether the model instance is A or B. You would want to pick `a` (or some subset of) for the features, and not `b` since it does not contribute to the prediction.\n", + "\n", + " y = f(a,b)\n", + "\n", + "The minimum dictionary format for `inputs` is:\n", + "\n", + " { \"inputs\": { \"[your_display_name]\":\n", + " \"input_tensor_name\": [layer]\n", + " }\n", + " }\n", + "\n", + "
\n", + " - [your_display_name]: A human readable name you assign to the input to explain. A common example is \"features\".
\n", + " - \"input_tensor_name\": The key/value field to identify the input layer for the feature attribution.
\n", + " - [layer]: The input layer for feature attribution. In a single input tensor model, it is the first (bottom-most) layer in the model.\n", + "
\n", + "\n", + "Since the inputs to the model are tabular, you can specify the following two additional fields as reporting/visualization aids:\n", + "\n", + "
\n", + " - \"modality\": \"image\": Indicates the field values are image data.\n", + "
\n", + "\n", + "Since the inputs to the model are tabular, you can specify the following two additional fields as reporting/visualization aids:\n", + "\n", + "
\n", + " - \"encoding\": \"BAG_OF_FEATURES\" : Indicates that the inputs are set of tabular features.
\n", + " - \"index_feature_mapping\": [ feature-names ] : A list of human readable names for each feature. For this example, we use the feature names specified in the dataset.
\n", + " - \"modality\": \"numeric\": Indicates the field values are numeric.\n", + "
" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "explanation_metadata:tabular" + }, + "outputs": [], + "source": [ + "INPUT_METADATA = {\n", + " \"input_tensor_name\": serving_input,\n", + " \"encoding\": \"BAG_OF_FEATURES\",\n", + " \"modality\": \"numeric\",\n", + " \"index_feature_mapping\": [\n", + " \"crim\",\n", + " \"zn\",\n", + " \"indus\",\n", + " \"chas\",\n", + " \"nox\",\n", + " \"rm\",\n", + " \"age\",\n", + " \"dis\",\n", + " \"rad\",\n", + " \"tax\",\n", + " \"ptratio\",\n", + " \"b\",\n", + " \"lstat\",\n", + " ],\n", + "}\n", + "\n", + "OUTPUT_METADATA = {\"output_tensor_name\": serving_output}\n", + "\n", + "input_metadata = aip.ExplanationMetadata.InputMetadata(INPUT_METADATA)\n", + "output_metadata = aip.ExplanationMetadata.OutputMetadata(OUTPUT_METADATA)\n", + "\n", + "metadata = aip.ExplanationMetadata(\n", + " inputs={\"features\": input_metadata}, outputs={\"medv\": output_metadata}\n", + ")\n", + "\n", + "explanation_spec = aip.ExplanationSpec(metadata=metadata, parameters=parameters)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "upload_the_model:explanation" + }, + "source": [ + "### Upload the model\n", + "\n", + "Use this helper function `upload_model` to upload your model, stored in SavedModel format, up to the `Model` service, which will instantiate a Vertex `Model` resource instance for your model. Once you've done that, you can use the `Model` resource instance in the same way as any other Vertex `Model` resource instance, such as deploying to an `Endpoint` resource for serving predictions.\n", + "\n", + "The helper function takes the following parameters:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` service.\n", + "- `image_uri`: The container image for the model deployment.\n", + "- `model_uri`: The Cloud Storage path to our SavedModel artificat. For this tutorial, this is the Cloud Storage location where the `trainer/task.py` saved the model artifacts, which we specified in the variable `MODEL_DIR`.\n", + "\n", + "The helper function calls the `Model` client service's method `upload_model`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for `Dataset`, `Model` and `Endpoint` resources.\n", + "- `model`: The specification for the Vertex `Model` resource instance.\n", + "\n", + "Let's now dive deeper into the Vertex model specification `model`. This is a dictionary object that consists of the following fields:\n", + "\n", + "- `display_name`: A human readable name for the `Model` resource.\n", + "- `metadata_schema_uri`: Since your model was built without an Vertex `Dataset` resource, you will leave this blank (`''`).\n", + "- `artificat_uri`: The Cloud Storage path where the model is stored in SavedModel format.\n", + "- `container_spec`: This is the specification for the Docker container that will be installed on the `Endpoint` resource, from which the `Model` resource will serve predictions. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + "- `explanation_spec`: This is the specification for enabling explainability for your model.\n", + "\n", + "Uploading a model into a Vertex Model resource returns a long running operation, since it may take a few moments. You call response.result(), which is a synchronous call and will return when the Vertex Model resource is ready.\n", + "\n", + "The helper function returns the Vertex fully qualified identifier for the corresponding Vertex Model instance upload_model_response.model. You will save the identifier for subsequent steps in the variable model_to_deploy_id." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "upload_the_model:explanation" + }, + "outputs": [], + "source": [ + "IMAGE_URI = DEPLOY_IMAGE\n", + "\n", + "\n", + "def upload_model(display_name, image_uri, model_uri):\n", + "\n", + " model = aip.Model(\n", + " display_name=display_name,\n", + " artifact_uri=model_uri,\n", + " metadata_schema_uri=\"\",\n", + " explanation_spec=explanation_spec,\n", + " container_spec={\"image_uri\": image_uri},\n", + " )\n", + "\n", + " response = clients[\"model\"].upload_model(parent=PARENT, model=model)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " upload_model_response = response.result(timeout=180)\n", + " print(\"upload_model_response\")\n", + " print(\" model:\", upload_model_response.model)\n", + " return upload_model_response.model\n", + "\n", + "\n", + "model_to_deploy_id = upload_model(\n", + " \"boston-\" + TIMESTAMP, IMAGE_URI, model_path_to_deploy\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_model" + }, + "source": [ + "### Get `Model` resource information\n", + "\n", + "Now let's get the model information for just your model. Use this helper function `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource.\n", + "\n", + "This helper function calls the Vertex `Model` client service's method `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_model" + }, + "outputs": [], + "source": [ + "def get_model(name):\n", + " response = clients[\"model\"].get_model(name=name)\n", + " print(response)\n", + "\n", + "\n", + "get_model(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:custom" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex custom `Model` resource. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"boston_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:dedicated,v1beta1" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the model to the endpoint you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified identifier of the `Model` resource to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to to.\n", + "- `deployed_model`: The requirements for deploying the model.\n", + "- `traffic_split`: Percent of traffic at endpoint that goes to this model, which is specified as a dictioney of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then specify as, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + " { \"0\": percent, model_id: percent, ... }\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified identifier of the (upload) `Model` resource to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `dedicated_resources`: This refers to how many compute instances (replicas) that are scaled for serving prediction requests.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `min_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "- `enable_container_logging`: This enables logging of container events, such as execution failures (default is container logging is disabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:dedicated,v1beta1" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"boston_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " \"machine_spec\": machine_spec,\n", + " },\n", + " \"enable_container_logging\": False,\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction:xai" + }, + "source": [ + "## Make a online prediction request with explainability\n", + "\n", + "Now do a online prediction with explainability to your deployed model. In this method, the predicted response will include an explanation on how the features contributed to the explanation." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_item:test" + }, + "source": [ + "### Get test item\n", + "\n", + "You will use an example out of the test (holdout) portion of the dataset as a test item." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:test,tabular" + }, + "outputs": [], + "source": [ + "test_item = x_test[0]\n", + "test_label = y_test[0]\n", + "print(test_item.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "send_explain_request:tabular" + }, + "source": [ + "### Send the prediction with explanation request\n", + "\n", + "Ok, now you have a test data item. Use this helper function `explain_data`, which takes the parameters:\n", + "\n", + "- `data_item`: The test data items as a numpy 1D array of floating point values.\n", + "- `endpoint`: The Vertex AI fully qualified identifier for the endpoint where the model was deployed.\n", + "- `parameters_dict`: Additional parameters for serving.\n", + "\n", + "This function uses the prediction client service and calls the `explain` method with the parameters:\n", + "\n", + "- `endpoint`: The Vertex AI fully qualified identifier for the endpoint where the model was deployed.\n", + "- `instances`: A list of instances (data items) to predict.\n", + "- `parameters`: Additional parameters for serving. *Note*, custom models do not support additional parameters.\n", + "- `deployed_model_id`: The Vertex AI fully qualified identifier for the deployed model, when more than one model is deployed at the endpoint. Otherwise, if only one model deployed, can be set to `None`.\n", + "\n", + "To pass the test data to the prediction service, you must package it for transmission to the serving binary as follows:\n", + "\n", + " 1. Convert the data item from a 1D numpy array to a 1D Python list.\n", + " 2. Convert the prediction request to a serialized Google protobuf (`json_format.ParseDict()`)\n", + "\n", + "\n", + "Each instance in the prediction request is a dictionary entry of the form:\n", + "\n", + " {input_name: content}\n", + "\n", + "- `input_name`: the name of the input layer of the underlying model.\n", + "- `content`: The data item as a 1D Python list.\n", + "\n", + "Since the `explain()` method can take multiple data items (instances), you will send your single data item as a list of one data item. As a final step, you package the instances list into Google's protobuf format -- which is what we pass to the `explain()` method.\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding image in the request. You will see in the output for each prediction:\n", + "\n", + "- `deployed_model_id` -- The Vertex AI fully qualified identifer for the model that did the prediction/explanation.\n", + "- `predictions` -- The predicated median value of a house in units of 1K USD.\n", + "- `explanations` -- How each feature contributed to the prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "send_explain_request:tabular" + }, + "outputs": [], + "source": [ + "def explain_data(\n", + " data_items, endpoint, parameters_dict, deployed_model_id, silent=False\n", + "):\n", + " parameters = json_format.ParseDict(parameters_dict, Value())\n", + "\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances = [\n", + " json_format.ParseDict({serving_input: s.tolist()}, Value()) for s in data_items\n", + " ]\n", + "\n", + " response = clients[\"prediction\"].explain(\n", + " endpoint=endpoint,\n", + " instances=instances,\n", + " parameters=parameters,\n", + " deployed_model_id=deployed_model_id,\n", + " )\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + "\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", prediction)\n", + "\n", + " explanations = response.explanations\n", + " print(\"explanations\")\n", + " for explanation in explanations:\n", + " print(explanation)\n", + " return response\n", + "\n", + "\n", + "response = explain_data([test_item], endpoint_id, None, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "understanding_explanations:boston" + }, + "source": [ + "### Understanding the explanations response\n", + "\n", + "First, let's look at the median house price your model predicted and compare it to the actual value." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "understanding_explanations:boston" + }, + "outputs": [], + "source": [ + "predictions = response.predictions\n", + "print(\"Predicted Median House Value:\", predictions[0][0], \"1K USD\")\n", + "print(\"Actual Mean House Value:\", test_label, \"1K USD\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "examine_feature_attributions" + }, + "source": [ + "### Examine feature attributions\n", + "\n", + "Next you will look at the feature attributions for this particular example. Positive attribution values mean a particular feature pushed your model prediction up by that amount, and vice versa for negative attribution values." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "examine_feature_attributions:boston" + }, + "outputs": [], + "source": [ + "from tabulate import tabulate\n", + "\n", + "feature_names = [\n", + " \"crim\",\n", + " \"zn\",\n", + " \"indus\",\n", + " \"chas\",\n", + " \"nox\",\n", + " \"rm\",\n", + " \"age\",\n", + " \"dis\",\n", + " \"rad\",\n", + " \"tax\",\n", + " \"ptratio\",\n", + " \"b\",\n", + " \"lstat\",\n", + "]\n", + "attributions = response.explanations[0].attributions[0].feature_attributions\n", + "\n", + "rows = []\n", + "for i, val in enumerate(feature_names):\n", + " rows.append([val, test_item[i], attributions[val]])\n", + "print(tabulate(rows, headers=[\"Feature name\", \"Feature value\", \"Attribution value\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "check_explanations_baselines" + }, + "source": [ + "### Check your explanations and baselines\n", + "\n", + "To better make sense of the feature attributions you're getting, you should compare them with your model's baseline. In most cases, the sum of your attribution values + the baseline should be very close to your model's predicted value for each input. Also note that for regression models, the `baseline_score` returned from AI Explanations will be the same for each example sent to your model. For classification models, each class will have its own baseline.\n", + "\n", + "In this section you'll send 10 test examples to your model for prediction in order to compare the feature attributions with the baseline. Then you'll run each test example's attributions through a sanity check in the `sanity_check_explanations` method.\n", + "\n", + "#### Get explanations" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "check_explanations_baselines:boston" + }, + "outputs": [], + "source": [ + "# Prepare 10 test examples to your model for prediction\n", + "pred_batch = []\n", + "for i in range(10):\n", + " pred_batch.append(x_test[i])\n", + "\n", + "response = explain_data(pred_batch, endpoint_id, None, None, silent=True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sanity_check_explanations" + }, + "source": [ + "#### Sanity check\n", + "\n", + "In the function below you perform a sanity check on the explanations." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "sanity_check_explanations" + }, + "outputs": [], + "source": [ + "def sanity_check_explanations(\n", + " explanation, prediction, mean_tgt_value=None, variance_tgt_value=None\n", + "):\n", + " passed_test = 0\n", + " total_test = 1\n", + " # `attributions` is a dict where keys are the feature names\n", + " # and values are the feature attributions for each feature\n", + " baseline_score = explanation.attributions[0].baseline_output_value\n", + " print(\"baseline:\", baseline_score)\n", + "\n", + " # Sanity check 1\n", + " # The prediction at the input is equal to that at the baseline.\n", + " # Please use a different baseline. Some suggestions are: random input, training\n", + " # set mean.\n", + " if abs(prediction - baseline_score) <= 0.05:\n", + " print(\"Warning: example score and baseline score are too close.\")\n", + " print(\"You might not get attributions.\")\n", + " else:\n", + " passed_test += 1\n", + " print(\"Sanity Check 1: Passed\")\n", + "\n", + " print(passed_test, \" out of \", total_test, \" sanity checks passed.\")\n", + "\n", + "\n", + "i = 0\n", + "for explanation in response.explanations:\n", + " try:\n", + " prediction = np.max(response.predictions[i][\"scores\"])\n", + " except TypeError:\n", + " prediction = np.max(response.predictions[i])\n", + " sanity_check_explanations(explanation, prediction)\n", + " i += 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_custom_tabular_regression_online_explain.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_custom_tabular_regression_online_pipeline.ipynb b/notebooks/community/gapic/custom/showcase_custom_tabular_regression_online_pipeline.ipynb new file mode 100644 index 000000000..dc1c3bbe8 --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_custom_tabular_regression_online_pipeline.ipynb @@ -0,0 +1,2187 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Custom training tabular regression model with pipeline for online prediction with training pipeline\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom,training_pipeline" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to train and deploy a custom tabular regression model for online prediction, using a training pipeline." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:custom,boston,lrg" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Boston Housing Prices dataset](https://www.cs.toronto.edu/~delve/data/boston/bostonDetail.html). The version of the dataset you will use in this tutorial is built into TensorFlow. The trained model predicts the median price of a house in units of 1K USD." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,training,online_prediction,training_pipeline" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create a custom model from a Python script in a Google prebuilt Docker container using the Vertex client library, and then do a prediction on the deployed model by sending data. You can alternatively create custom models using `gcloud` command-line tool or online using Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex custom job for training a model.\n", + "- Create a `TrainingPipeline` resource.\n", + "- Train a TensorFlow model with the `TrainingPipeline` resource.\n", + "- Retrieve and load the model artifacts.\n", + "- View the model evaluation.\n", + "- Upload the model as a Vertex `Model` resource.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "custom_constants" + }, + "source": [ + "#### CustomJob constants\n", + "\n", + "Set constants unique to CustomJob training:\n", + "\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "custom_constants" + }, + "outputs": [], + "source": [ + "CUSTOM_TASK_GCS_PATH = (\n", + " \"gs://google-cloud-aiplatform/schema/trainingjob/definition/custom_task_1.0.0.yaml\"\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for training and prediction.\n", + "\n", + "Set the variables `TRAIN_GPU/TRAIN_NGPU` and `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU.\n", + "\n", + "*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3 -- which is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n", + " TRAIN_GPU, TRAIN_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n", + " )\n", + "else:\n", + " TRAIN_GPU, TRAIN_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (None, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:training,prediction" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for training and prediction\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-3:latest`\n", + " - TensorFlow 2.4\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-4:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/training/xgboost-cpu.1-1`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/training/scikit-learn-cpu.0-23:latest`\n", + " - Pytorch\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-5:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-6:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-7:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for training](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers).\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-3:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-90:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-82:latest`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-23:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-22:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-20:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for prediction](https://cloud.google.com/vertex-ai/docs/predictions/pre-built-containers)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf2-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf2-cpu.{}\".format(TF)\n", + "else:\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "TRAIN_IMAGE = \"gcr.io/cloud-aiplatform/training/{}:latest\".format(TRAIN_VERSION)\n", + "DEPLOY_IMAGE = \"gcr.io/cloud-aiplatform/prediction/{}:latest\".format(DEPLOY_VERSION)\n", + "\n", + "print(\"Training:\", TRAIN_IMAGE, TRAIN_GPU, TRAIN_NGPU)\n", + "print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU, DEPLOY_NGPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:training,prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for training and prediction.\n", + "\n", + "- Set the variables `TRAIN_COMPUTE` and `DEPLOY_COMPUTE` to configure the compute resources for the VMs you will use for for training and prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: The following is not supported for training:*\n", + "\n", + " - `standard`: 2 vCPUs\n", + " - `highcpu`: 2, 4 and 8 vCPUs\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_TRAIN_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Train machine type\", TRAIN_COMPUTE)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own custom model and training for Boston Housing." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:custom,training_pipeline" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:custom,training_pipeline" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_model" + }, + "source": [ + "## Train a model\n", + "\n", + "There are two ways you can train a custom model using a container image:\n", + "\n", + "- **Use a Google Cloud prebuilt container**. If you use a prebuilt container, you will additionally specify a Python package to install into the container image. This Python package contains your code for training a custom model.\n", + "\n", + "- **Use your own custom container image**. If you use your own container, the container needs to contain your code for training a custom model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_specification:prebuilt_container" + }, + "source": [ + "## Prepare your custom job specification\n", + "\n", + "Now that your clients are ready, your first step is to create a Job Specification for your custom training job. The job specification will consist of the following:\n", + "\n", + "- `worker_pool_spec` : The specification of the type of machine(s) you will use for training and how many (single or distributed)\n", + "- `python_package_spec` : The specification of the Python package to be installed with the pre-built container." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "source": [ + "### Prepare your machine specification\n", + "\n", + "Now define the machine specification for your custom training job. This tells Vertex what type of machine instance to provision for the training.\n", + " - `machine_type`: The type of GCP instance to provision -- e.g., n1-standard-8.\n", + " - `accelerator_type`: The type, if any, of hardware accelerator. In this tutorial if you previously set the variable `TRAIN_GPU != None`, you are using a GPU; otherwise you will use a CPU.\n", + " - `accelerator_count`: The number of accelerators." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "outputs": [], + "source": [ + "if TRAIN_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": TRAIN_COMPUTE,\n", + " \"accelerator_type\": TRAIN_GPU,\n", + " \"accelerator_count\": TRAIN_NGPU,\n", + " }\n", + "else:\n", + " machine_spec = {\"machine_type\": TRAIN_COMPUTE, \"accelerator_count\": 0}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "source": [ + "### Prepare your disk specification\n", + "\n", + "(optional) Now define the disk specification for your custom training job. This tells Vertex what type and size of disk to provision in each machine instance for the training.\n", + "\n", + " - `boot_disk_type`: Either SSD or Standard. SSD is faster, and Standard is less expensive. Defaults to SSD.\n", + " - `boot_disk_size_gb`: Size of disk in GB." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "outputs": [], + "source": [ + "DISK_TYPE = \"pd-ssd\" # [ pd-ssd, pd-standard]\n", + "DISK_SIZE = 200 # GB\n", + "\n", + "disk_spec = {\"boot_disk_type\": DISK_TYPE, \"boot_disk_size_gb\": DISK_SIZE}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container,tabular" + }, + "source": [ + "### Define the worker pool specification\n", + "\n", + "Next, you define the worker pool specification for your custom training job. The worker pool specification will consist of the following:\n", + "\n", + "- `replica_count`: The number of instances to provision of this machine type.\n", + "- `machine_spec`: The hardware specification.\n", + "- `disk_spec` : (optional) The disk storage specification.\n", + "\n", + "- `python_package`: The Python training package to install on the VM instance(s) and which Python module to invoke, along with command line arguments for the Python module.\n", + "\n", + "Let's dive deeper now into the python package specification:\n", + "\n", + "-`executor_image_spec`: This is the docker image which is configured for your custom training job.\n", + "\n", + "-`package_uris`: This is a list of the locations (URIs) of your python training packages to install on the provisioned instance. The locations need to be in a Cloud Storage bucket. These can be either individual python files or a zip (archive) of an entire package. In the later case, the job service will unzip (unarchive) the contents into the docker image.\n", + "\n", + "-`python_module`: The Python module (script) to invoke for running the custom training job. In this example, you will be invoking `trainer.task.py` -- note that it was not neccessary to append the `.py` suffix.\n", + "\n", + "-`args`: The command line arguments to pass to the corresponding Pythom module. In this example, you will be setting:\n", + " - `\"--model-dir=\" + MODEL_DIR` : The Cloud Storage location where to store the model artifacts. There are two ways to tell the training script where to save the model artifacts:\n", + " - direct: You pass the Cloud Storage location as a command line argument to your training script (set variable `DIRECT = True`), or\n", + " - indirect: The service passes the Cloud Storage location as the environment variable `AIP_MODEL_DIR` to your training script (set variable `DIRECT = False`). In this case, you tell the service the model artifact location in the job specification.\n", + " - `\"--epochs=\" + EPOCHS`: The number of epochs for training.\n", + " - `\"--steps=\" + STEPS`: The number of steps (batches) per epoch.\n", + " - `\"--distribute=\" + TRAIN_STRATEGY\"` : The training distribution strategy to use for single or distributed training.\n", + " - `\"single\"`: single device.\n", + " - `\"mirror\"`: all GPU devices on a single compute instance.\n", + " - `\"multi\"`: all GPU devices on all compute instances.\n", + " - `\"--param-file=\" + PARAM_FILE`: The Cloud Storage location for storing feature normalization values." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container,tabular" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_job_\" + TIMESTAMP\n", + "MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, JOB_NAME)\n", + "\n", + "if not TRAIN_NGPU or TRAIN_NGPU < 2:\n", + " TRAIN_STRATEGY = \"single\"\n", + "else:\n", + " TRAIN_STRATEGY = \"mirror\"\n", + "\n", + "EPOCHS = 20\n", + "STEPS = 100\n", + "\n", + "PARAM_FILE = BUCKET_NAME + \"/params.txt\"\n", + "\n", + "DIRECT = True\n", + "if DIRECT:\n", + " CMDARGS = [\n", + " \"--model-dir=\" + MODEL_DIR,\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " \"--param-file=\" + PARAM_FILE,\n", + " ]\n", + "else:\n", + " CMDARGS = [\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " \"--param-file=\" + PARAM_FILE,\n", + " ]\n", + "\n", + "worker_pool_spec = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": machine_spec,\n", + " \"disk_spec\": disk_spec,\n", + " \"python_package_spec\": {\n", + " \"executor_image_uri\": TRAIN_IMAGE,\n", + " \"package_uris\": [BUCKET_NAME + \"/trainer_boston.tar.gz\"],\n", + " \"python_module\": \"trainer.task\",\n", + " \"args\": CMDARGS,\n", + " },\n", + " }\n", + "]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "examine_training_package" + }, + "source": [ + "### Examine the training package\n", + "\n", + "#### Package layout\n", + "\n", + "Before you start the training, you will look at how a Python package is assembled for a custom training job. When unarchived, the package contains the following directory/file layout.\n", + "\n", + "- PKG-INFO\n", + "- README.md\n", + "- setup.cfg\n", + "- setup.py\n", + "- trainer\n", + " - \\_\\_init\\_\\_.py\n", + " - task.py\n", + "\n", + "The files `setup.cfg` and `setup.py` are the instructions for installing the package into the operating environment of the Docker image.\n", + "\n", + "The file `trainer/task.py` is the Python script for executing the custom training job. *Note*, when we referred to it in the worker pool specification, we replace the directory slash with a dot (`trainer.task`) and dropped the file suffix (`.py`).\n", + "\n", + "#### Package Assembly\n", + "\n", + "In the following cells, you will assemble the training package." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "examine_training_package" + }, + "outputs": [], + "source": [ + "# Make folder for Python training script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\ntag_build =\\n\\ntag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'tensorflow_datasets==1.3.0',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\nName: Boston Housing tabular regression\\n\\nVersion: 0.0.0\\n\\nSummary: Demostration training script\\n\\nHome-page: www.google.com\\n\\nAuthor: Google\\n\\nAuthor-email: aferlitsch@google.com\\n\\nLicense: Public\\n\\nDescription: Demo\\n\\nPlatform: Vertex\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:boston" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the training script task.py. I won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n", + "- Loads Boston Housing dataset from TF.Keras builtin datasets\n", + "- Builds a simple deep neural network model using TF.Keras model API.\n", + "- Compiles the model (`compile()`).\n", + "- Sets a training distribution strategy according to the argument `args.distribute`.\n", + "- Trains the model (`fit()`) with epochs specified by `args.epochs`.\n", + "- Saves the trained model (`save(args.model_dir)`) to the specified model directory.\n", + "- Saves the maximum value for each feature `f.write(str(params))` to the specified parameters file." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:boston" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Single, Mirror and Multi-Machine Distributed Training for Boston Housing\n", + "\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "import numpy as np\n", + "import argparse\n", + "import os\n", + "import sys\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv('AIP_MODEL_DIR'), type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.001, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=20, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=100, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "parser.add_argument('--param-file', dest='param_file',\n", + " default='/tmp/param.txt', type=str,\n", + " help='Output file for parameters')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "\n", + "# Single Machine, single compute device\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "# Single Machine, multiple compute device\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "# Multiple Machine, multiple compute device\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "# Multi-worker configuration\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "\n", + "def make_dataset():\n", + "\n", + " # Scaling Boston Housing data features\n", + " def scale(feature):\n", + " max = np.max(feature)\n", + " feature = (feature / max).astype(np.float)\n", + " return feature, max\n", + "\n", + " (x_train, y_train), (x_test, y_test) = tf.keras.datasets.boston_housing.load_data(\n", + " path=\"boston_housing.npz\", test_split=0.2, seed=113\n", + " )\n", + " params = []\n", + " for _ in range(13):\n", + " x_train[_], max = scale(x_train[_])\n", + " x_test[_], _ = scale(x_test[_])\n", + " params.append(max)\n", + "\n", + " # store the normalization (max) value for each feature\n", + " with tf.io.gfile.GFile(args.param_file, 'w') as f:\n", + " f.write(str(params))\n", + " return (x_train, y_train), (x_test, y_test)\n", + "\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_dnn_model():\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Dense(128, activation='relu', input_shape=(13,)),\n", + " tf.keras.layers.Dense(128, activation='relu'),\n", + " tf.keras.layers.Dense(1, activation='linear')\n", + " ])\n", + " model.compile(\n", + " loss='mse',\n", + " optimizer=tf.keras.optimizers.RMSprop(learning_rate=args.lr))\n", + " return model\n", + "\n", + "NUM_WORKERS = strategy.num_replicas_in_sync\n", + "# Here the batch size scales up by number of workers since\n", + "# `tf.data.Dataset.batch` expects the global batch size.\n", + "BATCH_SIZE = 16\n", + "GLOBAL_BATCH_SIZE = BATCH_SIZE * NUM_WORKERS\n", + "\n", + "with strategy.scope():\n", + " # Creation of dataset, and model building/compiling need to be within\n", + " # `strategy.scope()`.\n", + " model = build_and_compile_dnn_model()\n", + "\n", + "# Train the model\n", + "(x_train, y_train), (x_test, y_test) = make_dataset()\n", + "model.fit(x_train, y_train, epochs=args.epochs, batch_size=GLOBAL_BATCH_SIZE)\n", + "model.save(args.model_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tarball_training_script" + }, + "source": [ + "#### Store training script on your Cloud Storage bucket\n", + "\n", + "Next, you package the training folder into a compressed tar ball, and then store it in your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tarball_training_script" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_boston.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_with_pipeline" + }, + "source": [ + "## Train the model using a `TrainingPipeline` resource\n", + "\n", + "Now start training of your custom training job using a training pipeline on Vertex. To train the your custom model, do the following steps:\n", + "\n", + "1. Create a Vertex `TrainingPipeline` resource for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training.\n", + "\n", + "### Create a `TrainingPipeline` resource\n", + "\n", + "You may ask, what do we use a pipeline for? We typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "#### The `training_pipeline` specification\n", + "\n", + "First, you need to describe a pipeline specification. Let's look into the *minimal* requirements for constructing a `training_pipeline` specification for a custom job:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The training task schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A dictionary describing the specification for the (uploaded) Vertex custom `Model` resource.\n", + " - `display_name`: A human readable name for the `Model` resource.\n", + " - `artificat_uri`: The Cloud Storage path where the model artifacts are stored in SavedModel format.\n", + " - `container_spec`: This is the specification for the Docker container that will be installed on the `Endpoint` resource, from which the custom model will serve predictions." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_with_pipeline" + }, + "outputs": [], + "source": [ + "from google.protobuf import json_format\n", + "from google.protobuf.struct_pb2 import Value\n", + "\n", + "MODEL_NAME = \"custom_pipeline-\" + TIMESTAMP\n", + "PIPELINE_DISPLAY_NAME = \"custom-training-pipeline\" + TIMESTAMP\n", + "\n", + "training_task_inputs = json_format.ParseDict(\n", + " {\"workerPoolSpecs\": worker_pool_spec}, Value()\n", + ")\n", + "pipeline = {\n", + " \"display_name\": PIPELINE_DISPLAY_NAME,\n", + " \"training_task_definition\": CUSTOM_TASK_GCS_PATH,\n", + " \"training_task_inputs\": training_task_inputs,\n", + " \"model_to_upload\": {\n", + " \"display_name\": PIPELINE_DISPLAY_NAME + \"-model\",\n", + " \"artifact_uri\": MODEL_DIR,\n", + " \"container_spec\": {\"image_uri\": DEPLOY_IMAGE},\n", + " },\n", + "}\n", + "\n", + "print(pipeline)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_training_pipeline:custom" + }, + "source": [ + "#### Create the training pipeline\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameter:\n", + "\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "The helper function calls the pipeline client service's `create_pipeline` method, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: The full specification for the pipeline training job.\n", + "\n", + "The helper function will return the Vertex fully qualified identifier assigned to the training pipeline, which is saved as `pipeline.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_training_pipeline:custom" + }, + "outputs": [], + "source": [ + "def create_pipeline(training_pipeline):\n", + "\n", + " try:\n", + " pipeline = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " )\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline\n", + "\n", + "\n", + "response = create_pipeline(pipeline)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "model_path_to_deploy" + }, + "outputs": [], + "source": [ + "if not DIRECT:\n", + " MODEL_DIR = MODEL_DIR + \"/model\"\n", + "model_path_to_deploy = MODEL_DIR" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "load_saved_model" + }, + "source": [ + "## Load the saved model\n", + "\n", + "Your model is stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Now load it from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n", + "\n", + "To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "load_saved_model" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:tabular" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now let's find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load the Boston Housing test (holdout) data from `tf.keras.datasets`, using the method `load_data()`. This will return the dataset as a tuple of two elements. The first element is the training data and the second is the test data. Each element is also a tuple of two elements: the feature data, and the corresponding labels (median value of owner-occupied home).\n", + "\n", + "You don't need the training data, and hence why we loaded it as `(_, _)`.\n", + "\n", + "Before you can run the data through evaluation, you need to preprocess it:\n", + "\n", + "x_test:\n", + "1. Normalize (rescaling) the data in each column by dividing each value by the maximum value of that column. This will replace each single value with a 32-bit floating point number between 0 and 1." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "evaluate_custom_model:tabular,boston" + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "from tensorflow.keras.datasets import boston_housing\n", + "\n", + "(_, _), (x_test, y_test) = boston_housing.load_data(\n", + " path=\"boston_housing.npz\", test_split=0.2, seed=113\n", + ")\n", + "\n", + "\n", + "def scale(feature):\n", + " max = np.max(feature)\n", + " feature = (feature / max).astype(np.float32)\n", + " return feature\n", + "\n", + "\n", + "# Let's save one data item that has not been scaled\n", + "x_test_notscaled = x_test[0:1].copy()\n", + "\n", + "for _ in range(13):\n", + " x_test[_] = scale(x_test[_])\n", + "x_test = x_test.astype(np.float32)\n", + "\n", + "print(x_test.shape, x_test.dtype, y_test.shape)\n", + "print(\"scaled\", x_test[0])\n", + "print(\"unscaled\", x_test_notscaled)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "perform_evaluation_custom" + }, + "source": [ + "### Perform the model evaluation\n", + "\n", + "Now evaluate how well the model in the custom job did." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "perform_evaluation_custom" + }, + "outputs": [], + "source": [ + "model.evaluate(x_test, y_test)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "how_serving_function_works" + }, + "source": [ + "## Upload the model for serving\n", + "\n", + "Next, you will upload your TF.Keras model from the custom job to Vertex `Model` service, which will create a Vertex `Model` resource for your custom model. During upload, you need to define a serving function to convert data to the format your model expects. If you send encoded data to Vertex, your serving function ensures that the data is decoded on the model server before it is passed as input to your model.\n", + "\n", + "### How does the serving function work\n", + "\n", + "When you send a request to an online prediction server, the request is received by a HTTP server. The HTTP server extracts the prediction request from the HTTP request content body. The extracted prediction request is forwarded to the serving function. For Google pre-built prediction containers, the request content is passed to the serving function as a `tf.string`.\n", + "\n", + "The serving function consists of two parts:\n", + "\n", + "- `preprocessing function`:\n", + " - Converts the input (`tf.string`) to the input shape and data type of the underlying model (dynamic graph).\n", + " - Performs the same preprocessing of the data that was done during training the underlying model -- e.g., normalizing, scaling, etc.\n", + "- `post-processing function`:\n", + " - Converts the model output to format expected by the receiving application -- e.q., compresses the output.\n", + " - Packages the output for the the receiving application -- e.g., add headings, make JSON object, etc.\n", + "\n", + "Both the preprocessing and post-processing functions are converted to static graphs which are fused to the model. The output from the underlying model is passed to the post-processing function. The post-processing function passes the converted/packaged output back to the HTTP server. The HTTP server returns the output as the HTTP response content.\n", + "\n", + "One consideration you need to consider when building serving functions for TF.Keras models is that they run as static graphs. That means, you cannot use TF graph operations that require a dynamic graph. If you do, you will get an error during the compile of the serving function which will indicate that you are using an EagerTensor which is not supported." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_signature" + }, + "source": [ + "## Get the serving function signature\n", + "\n", + "You can get the signatures of your model's input and output layers by reloading the model into memory, and querying it for the signatures corresponding to each layer.\n", + "\n", + "When making a prediction request, you need to route the request to the serving function instead of the model, so you need to know the input layer name of the serving function -- which you will use later when you make a prediction request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_signature" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_path_to_deploy)\n", + "\n", + "serving_input = list(\n", + " loaded.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "print(\"Serving function input:\", serving_input)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "upload_the_model" + }, + "source": [ + "### Upload the model\n", + "\n", + "Use this helper function `upload_model` to upload your model, stored in SavedModel format, up to the `Model` service, which will instantiate a Vertex `Model` resource instance for your model. Once you've done that, you can use the `Model` resource instance in the same way as any other Vertex `Model` resource instance, such as deploying to an `Endpoint` resource for serving predictions.\n", + "\n", + "The helper function takes the following parameters:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` service.\n", + "- `image_uri`: The container image for the model deployment.\n", + "- `model_uri`: The Cloud Storage path to our SavedModel artificat. For this tutorial, this is the Cloud Storage location where the `trainer/task.py` saved the model artifacts, which we specified in the variable `MODEL_DIR`.\n", + "\n", + "The helper function calls the `Model` client service's method `upload_model`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for `Dataset`, `Model` and `Endpoint` resources.\n", + "- `model`: The specification for the Vertex `Model` resource instance.\n", + "\n", + "Let's now dive deeper into the Vertex model specification `model`. This is a dictionary object that consists of the following fields:\n", + "\n", + "- `display_name`: A human readable name for the `Model` resource.\n", + "- `metadata_schema_uri`: Since your model was built without an Vertex `Dataset` resource, you will leave this blank (`''`).\n", + "- `artificat_uri`: The Cloud Storage path where the model is stored in SavedModel format.\n", + "- `container_spec`: This is the specification for the Docker container that will be installed on the `Endpoint` resource, from which the `Model` resource will serve predictions. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + "\n", + "Uploading a model into a Vertex Model resource returns a long running operation, since it may take a few moments. You call response.result(), which is a synchronous call and will return when the Vertex Model resource is ready.\n", + "\n", + "The helper function returns the Vertex fully qualified identifier for the corresponding Vertex Model instance upload_model_response.model. You will save the identifier for subsequent steps in the variable model_to_deploy_id." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "upload_the_model" + }, + "outputs": [], + "source": [ + "IMAGE_URI = DEPLOY_IMAGE\n", + "\n", + "\n", + "def upload_model(display_name, image_uri, model_uri):\n", + " model = {\n", + " \"display_name\": display_name,\n", + " \"metadata_schema_uri\": \"\",\n", + " \"artifact_uri\": model_uri,\n", + " \"container_spec\": {\n", + " \"image_uri\": image_uri,\n", + " \"command\": [],\n", + " \"args\": [],\n", + " \"env\": [{\"name\": \"env_name\", \"value\": \"env_value\"}],\n", + " \"ports\": [{\"container_port\": 8080}],\n", + " \"predict_route\": \"\",\n", + " \"health_route\": \"\",\n", + " },\n", + " }\n", + " response = clients[\"model\"].upload_model(parent=PARENT, model=model)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " upload_model_response = response.result(timeout=180)\n", + " print(\"upload_model_response\")\n", + " print(\" model:\", upload_model_response.model)\n", + " return upload_model_response.model\n", + "\n", + "\n", + "model_to_deploy_id = upload_model(\n", + " \"boston-\" + TIMESTAMP, IMAGE_URI, model_path_to_deploy\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_model" + }, + "source": [ + "### Get `Model` resource information\n", + "\n", + "Now let's get the model information for just your model. Use this helper function `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource.\n", + "\n", + "This helper function calls the Vertex `Model` client service's method `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_model" + }, + "outputs": [], + "source": [ + "def get_model(name):\n", + " response = clients[\"model\"].get_model(name=name)\n", + " print(response)\n", + "\n", + "\n", + "get_model(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:custom" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex custom `Model` resource. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"boston_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:dedicated" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `dedicated_resources`: This refers to how many compute instances (replicas) that are scaled for serving prediction requests.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `min_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:dedicated" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"boston_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " \"machine_spec\": machine_spec,\n", + " },\n", + " \"disable_container_logging\": False,\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_item:test" + }, + "source": [ + "### Get test item\n", + "\n", + "You will use an example out of the test (holdout) portion of the dataset as a test item." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:test,tabular" + }, + "outputs": [], + "source": [ + "test_item = x_test[0]\n", + "test_label = y_test[0]\n", + "print(test_item.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "send_prediction_request:tabular" + }, + "source": [ + "### Send the prediction request\n", + "\n", + "Ok, now you have a test data item. Use this helper function `predict_data`, which takes the parameters:\n", + "\n", + "- `data`: The test data item as a numpy 1D array of floating point values.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `parameters_dict`: Additional parameters for serving.\n", + "\n", + "This function uses the prediction client service and calls the `predict` method with the parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `instances`: A list of instances (data items) to predict.\n", + "- `parameters`: Additional parameters for serving.\n", + "\n", + "To pass the test data to the prediction service, you package it for transmission to the serving binary as follows:\n", + "\n", + " 1. Convert the data item from a 1D numpy array to a 1D Python list.\n", + " 2. Convert the prediction request to a serialized Google protobuf (`json_format.ParseDict()`)\n", + "\n", + "\n", + "Each instance in the prediction request is a dictionary entry of the form:\n", + "\n", + " {input_name: content}\n", + "\n", + "- `input_name`: the name of the input layer of the underlying model.\n", + "- `content`: The data item as a 1D Python list.\n", + "\n", + "Since the `predict()` service can take multiple data items (instances), you will send your single data item as a list of one data item. As a final step, you package the instances list into Google's protobuf format -- which is what we pass to the `predict()` service.\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding image in the request. You will see in the output for each prediction:\n", + "\n", + "- `predictions` -- the predicated median value of a house in units of 1K USD." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "send_prediction_request:tabular" + }, + "outputs": [], + "source": [ + "def predict_data(data, endpoint, parameters_dict):\n", + " parameters = json_format.ParseDict(parameters_dict, Value())\n", + "\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [{serving_input: data.tolist()}]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", prediction)\n", + "\n", + "\n", + "predict_data(test_item, endpoint_id, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_custom_tabular_regression_online_pipeline.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_custom_tabular_regression_online_tfserving.ipynb b/notebooks/community/gapic/custom/showcase_custom_tabular_regression_online_tfserving.ipynb new file mode 100644 index 000000000..8d5bccd27 --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_custom_tabular_regression_online_tfserving.ipynb @@ -0,0 +1,2190 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Custom training tabular regression model using TF Serving container for online prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom,training,tfserving" + }, + "source": [ + "## Overview\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to train and deploy a custom tabular regression model for online prediction, using TF Serving container for prediction." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:custom,boston,lrg" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Boston Housing Prices dataset](https://www.cs.toronto.edu/~delve/data/boston/bostonDetail.html). The version of the dataset you will use in this tutorial is built into TensorFlow. The trained model predicts the median price of a house in units of 1K USD." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,training,online_prediction,tfserving" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial you create a custom model from a Python script in a Google prebuilt Docker container using the Vertex client library, and then do a prediction on the deployed model using TF Serving container. You can alternatively create custom models using `gcloud` command-line tool or online using Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex custom job for training a model.\n", + "- Create a `TrainingPipeline` resource.\n", + "- Train a TensorFlow model.\n", + "- Retrieve and load the model artifacts.\n", + "- View the model evaluation.\n", + "- Upload the model as a Vertex `Model` resource.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource with a TF Serving container.\n", + "- Make a prediction.\n", + "- Undeploy the `Model` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_docker" + }, + "source": [ + "### Install Docker (Colab or Local)\n", + "\n", + "If you are using Google Cloud Notebook, Docker is already installed. Skip these steps.\n", + "\n", + "By default, Docker is not installed on Colab. If you're running colab, then you need to do the following to install Docker." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_docker" + }, + "outputs": [], + "source": [ + "if \"google.colab\" in sys.modules:\n", + " ! sudo apt update\n", + " ! sudo apt install apt-transport-https ca-certificates curl software-properties-common\n", + " ! curl -fsSL https://download.docker.com/linux/ubuntu/gpg | sudo apt-key add -\n", + " ! sudo add-apt-repository \"deb [arch=amd64] https://download.docker.com/linux/ubuntu bionic stable\"\n", + " ! sudo apt update\n", + " ! sudo apt install docker-ce" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "start_docker_service" + }, + "source": [ + "#### Start Docker service\n", + "\n", + "Start the `docker` service." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "start_docker_service" + }, + "outputs": [], + "source": [ + "if \"google.colab\" in sys.modules:\n", + " ! sudo service docker start" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for training and prediction.\n", + "\n", + "Set the variables `TRAIN_GPU/TRAIN_NGPU` and `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU.\n", + "\n", + "*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3 -- which is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n", + " TRAIN_GPU, TRAIN_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n", + " )\n", + "else:\n", + " TRAIN_GPU, TRAIN_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (None, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:training" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for training.\n", + "\n", + "- Set the variable `TF` to the TensorFlow version of the container image. For example, `2-1` would be version 2.1, and `1-15` would be version 1.15. The following list shows some of the pre-built images available:\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-3:latest`\n", + " - TensorFlow 2.4\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-4:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/training/xgboost-cpu.1-1`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/training/scikit-learn-cpu.0-23:latest`\n", + " - Pytorch\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-5:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-6:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-7:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for training](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:training" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + "else:\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "TRAIN_IMAGE = \"gcr.io/cloud-aiplatform/training/{}:latest\".format(TRAIN_VERSION)\n", + "\n", + "print(\"Training:\", TRAIN_IMAGE, TRAIN_GPU, TRAIN_NGPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:tfserving" + }, + "source": [ + "#### Container (Docker) image for prediction\n", + "\n", + "Next, you set the TF Serving Docker container image for prediction.\n", + "\n", + " 1. Pull the corresponding CPU or GPU Docker image for TF Serving from Docker Hub.\n", + " 2. Create a tag for registering the image with Cloud Container Registry (gcr.io)\n", + " 3. Register the image with Cloud Container Registry.\n", + "\n", + "For more details on using TF Serving container, see [TF Serving](https://www.tensorflow.org/tfx/serving/docker)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:tfserving" + }, + "outputs": [], + "source": [ + "if DEPLOY_GPU:\n", + " ! docker pull tensorflow/serving:latest-gpu\n", + " DEPLOY_IMAGE = \"gcr.io/\" + PROJECT_ID + \"/tf_serving:gpu\"\n", + "else:\n", + " ! docker pull tensorflow/serving:latest\n", + " DEPLOY_IMAGE = \"gcr.io/\" + PROJECT_ID + \"/tf_serving\"\n", + "\n", + "! docker tag tensorflow/serving $DEPLOY_IMAGE\n", + "! docker push $DEPLOY_IMAGE\n", + "\n", + "print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU, DEPLOY_NGPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:training,prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for training and prediction.\n", + "\n", + "- Set the variables `TRAIN_COMPUTE` and `DEPLOY_COMPUTE` to configure the compute resources for the VMs you will use for for training and prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: The following is not supported for training:*\n", + "\n", + " - `standard`: 2 vCPUs\n", + " - `highcpu`: 2, 4 and 8 vCPUs\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_TRAIN_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Train machine type\", TRAIN_COMPUTE)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own custom model and training for Boston Housing." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:custom" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Model Service for `Model` resources.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:custom" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"job\"] = create_job_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_model" + }, + "source": [ + "## Train a model\n", + "\n", + "There are two ways you can train a custom model using a container image:\n", + "\n", + "- **Use a Google Cloud prebuilt container**. If you use a prebuilt container, you will additionally specify a Python package to install into the container image. This Python package contains your code for training a custom model.\n", + "\n", + "- **Use your own custom container image**. If you use your own container, the container needs to contain your code for training a custom model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_specification:prebuilt_container" + }, + "source": [ + "## Prepare your custom job specification\n", + "\n", + "Now that your clients are ready, your first step is to create a Job Specification for your custom training job. The job specification will consist of the following:\n", + "\n", + "- `worker_pool_spec` : The specification of the type of machine(s) you will use for training and how many (single or distributed)\n", + "- `python_package_spec` : The specification of the Python package to be installed with the pre-built container." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "source": [ + "### Prepare your machine specification\n", + "\n", + "Now define the machine specification for your custom training job. This tells Vertex what type of machine instance to provision for the training.\n", + " - `machine_type`: The type of GCP instance to provision -- e.g., n1-standard-8.\n", + " - `accelerator_type`: The type, if any, of hardware accelerator. In this tutorial if you previously set the variable `TRAIN_GPU != None`, you are using a GPU; otherwise you will use a CPU.\n", + " - `accelerator_count`: The number of accelerators." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "outputs": [], + "source": [ + "if TRAIN_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": TRAIN_COMPUTE,\n", + " \"accelerator_type\": TRAIN_GPU,\n", + " \"accelerator_count\": TRAIN_NGPU,\n", + " }\n", + "else:\n", + " machine_spec = {\"machine_type\": TRAIN_COMPUTE, \"accelerator_count\": 0}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "source": [ + "### Prepare your disk specification\n", + "\n", + "(optional) Now define the disk specification for your custom training job. This tells Vertex what type and size of disk to provision in each machine instance for the training.\n", + "\n", + " - `boot_disk_type`: Either SSD or Standard. SSD is faster, and Standard is less expensive. Defaults to SSD.\n", + " - `boot_disk_size_gb`: Size of disk in GB." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "outputs": [], + "source": [ + "DISK_TYPE = \"pd-ssd\" # [ pd-ssd, pd-standard]\n", + "DISK_SIZE = 200 # GB\n", + "\n", + "disk_spec = {\"boot_disk_type\": DISK_TYPE, \"boot_disk_size_gb\": DISK_SIZE}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container,tfserving" + }, + "source": [ + "### Define the worker pool specification\n", + "\n", + "Next, you define the worker pool specification for your custom training job. The worker pool specification will consist of the following:\n", + "\n", + "- `replica_count`: The number of instances to provision of this machine type.\n", + "- `machine_spec`: The hardware specification.\n", + "- `disk_spec` : (optional) The disk storage specification.\n", + "\n", + "- `python_package`: The Python training package to install on the VM instance(s) and which Python module to invoke, along with command line arguments for the Python module.\n", + "\n", + "Let's dive deeper now into the python package specification:\n", + "\n", + "-`executor_image_spec`: This is the docker image which is configured for your custom training job.\n", + "\n", + "-`package_uris`: This is a list of the locations (URIs) of your python training packages to install on the provisioned instance. The locations need to be in a Cloud Storage bucket. These can be either individual python files or a zip (archive) of an entire package. In the later case, the job service will unzip (unarchive) the contents into the docker image.\n", + "\n", + "-`python_module`: The Python module (script) to invoke for running the custom training job. In this example, you will be invoking `trainer.task.py` -- note that it was not neccessary to append the `.py` suffix.\n", + "\n", + "-`args`: The command line arguments to pass to the corresponding Pythom module. In this example, you will be setting:\n", + " - `\"--model-dir=\" + MODEL_DIR` : The Cloud Storage location where to store the model artifacts. There are two ways to tell the training script where to save the model artifacts:\n", + " - direct: You pass the Cloud Storage location as a command line argument to your training script (set variable `DIRECT = True`), or\n", + " - indirect: The service passes the Cloud Storage location as the environment variable `AIP_MODEL_DIR` to your training script (set variable `DIRECT = False`). In this case, you tell the service the model artifact location in the job specification.\n", + " - `\"--epochs=\" + EPOCHS`: The number of epochs for training.\n", + " - `\"--steps=\" + STEPS`: The number of steps (batches) per epoch.\n", + " - `\"--distribute=\" + TRAIN_STRATEGY\"` : The training distribution strategy to use for single or distributed training.\n", + " - `\"single\"`: single device.\n", + " - `\"mirror\"`: all GPU devices on a single compute instance.\n", + " - `\"multi\"`: all GPU devices on all compute instances.\n", + "\n", + "*Note*: For TF Serving, the `MODEL_DIR` must end in a subfolder that is a number, e.g., 1." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container,tfserving" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_job_\" + TIMESTAMP\n", + "MODEL_DIR = \"{}/{}/1\".format(BUCKET_NAME, JOB_NAME)\n", + "\n", + "if not TRAIN_NGPU or TRAIN_NGPU < 2:\n", + " TRAIN_STRATEGY = \"single\"\n", + "else:\n", + " TRAIN_STRATEGY = \"mirror\"\n", + "\n", + "EPOCHS = 20\n", + "STEPS = 100\n", + "\n", + "DIRECT = True\n", + "if DIRECT:\n", + " CMDARGS = [\n", + " \"--model-dir=\" + MODEL_DIR,\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "else:\n", + " CMDARGS = [\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "\n", + "worker_pool_spec = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": machine_spec,\n", + " \"disk_spec\": disk_spec,\n", + " \"python_package_spec\": {\n", + " \"executor_image_uri\": TRAIN_IMAGE,\n", + " \"package_uris\": [BUCKET_NAME + \"/trainer_boston.tar.gz\"],\n", + " \"python_module\": \"trainer.task\",\n", + " \"args\": CMDARGS,\n", + " },\n", + " }\n", + "]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "assemble_custom_job_specification" + }, + "source": [ + "### Assemble a job specification\n", + "\n", + "Now assemble the complete description for the custom job specification:\n", + "\n", + "- `display_name`: The human readable name you assign to this custom job.\n", + "- `job_spec`: The specification for the custom job.\n", + " - `worker_pool_specs`: The specification for the machine VM instances.\n", + " - `base_output_directory`: This tells the service the Cloud Storage location where to save the model artifacts (when variable `DIRECT = False`). The service will then pass the location to the training script as the environment variable `AIP_MODEL_DIR`, and the path will be of the form:\n", + "\n", + " /model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "assemble_custom_job_specification" + }, + "outputs": [], + "source": [ + "if DIRECT:\n", + " job_spec = {\"worker_pool_specs\": worker_pool_spec}\n", + "else:\n", + " job_spec = {\n", + " \"worker_pool_specs\": worker_pool_spec,\n", + " \"base_output_directory\": {\"output_uri_prefix\": MODEL_DIR},\n", + " }\n", + "\n", + "custom_job = {\"display_name\": JOB_NAME, \"job_spec\": job_spec}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "examine_training_package" + }, + "source": [ + "### Examine the training package\n", + "\n", + "#### Package layout\n", + "\n", + "Before you start the training, you will look at how a Python package is assembled for a custom training job. When unarchived, the package contains the following directory/file layout.\n", + "\n", + "- PKG-INFO\n", + "- README.md\n", + "- setup.cfg\n", + "- setup.py\n", + "- trainer\n", + " - \\_\\_init\\_\\_.py\n", + " - task.py\n", + "\n", + "The files `setup.cfg` and `setup.py` are the instructions for installing the package into the operating environment of the Docker image.\n", + "\n", + "The file `trainer/task.py` is the Python script for executing the custom training job. *Note*, when we referred to it in the worker pool specification, we replace the directory slash with a dot (`trainer.task`) and dropped the file suffix (`.py`).\n", + "\n", + "#### Package Assembly\n", + "\n", + "In the following cells, you will assemble the training package." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "examine_training_package" + }, + "outputs": [], + "source": [ + "# Make folder for Python training script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\ntag_build =\\n\\ntag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'tensorflow_datasets==1.3.0',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\nName: Boston Housing tabular regression\\n\\nVersion: 0.0.0\\n\\nSummary: Demostration training script\\n\\nHome-page: www.google.com\\n\\nAuthor: Google\\n\\nAuthor-email: aferlitsch@google.com\\n\\nLicense: Public\\n\\nDescription: Demo\\n\\nPlatform: Vertex\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:boston" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the training script task.py. I won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n", + "- Loads Boston Housing dataset from TF.Keras builtin datasets\n", + "- Builds a simple deep neural network model using TF.Keras model API.\n", + "- Compiles the model (`compile()`).\n", + "- Sets a training distribution strategy according to the argument `args.distribute`.\n", + "- Trains the model (`fit()`) with epochs specified by `args.epochs`.\n", + "- Saves the trained model (`save(args.model_dir)`) to the specified model directory.\n", + "- Saves the maximum value for each feature `f.write(str(params))` to the specified parameters file." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:boston" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Single, Mirror and Multi-Machine Distributed Training for Boston Housing\n", + "\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "import numpy as np\n", + "import argparse\n", + "import os\n", + "import sys\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv('AIP_MODEL_DIR'), type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.001, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=20, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=100, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "parser.add_argument('--param-file', dest='param_file',\n", + " default='/tmp/param.txt', type=str,\n", + " help='Output file for parameters')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "\n", + "# Single Machine, single compute device\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "# Single Machine, multiple compute device\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "# Multiple Machine, multiple compute device\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "# Multi-worker configuration\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "\n", + "def make_dataset():\n", + "\n", + " # Scaling Boston Housing data features\n", + " def scale(feature):\n", + " max = np.max(feature)\n", + " feature = (feature / max).astype(np.float)\n", + " return feature, max\n", + "\n", + " (x_train, y_train), (x_test, y_test) = tf.keras.datasets.boston_housing.load_data(\n", + " path=\"boston_housing.npz\", test_split=0.2, seed=113\n", + " )\n", + " params = []\n", + " for _ in range(13):\n", + " x_train[_], max = scale(x_train[_])\n", + " x_test[_], _ = scale(x_test[_])\n", + " params.append(max)\n", + "\n", + " # store the normalization (max) value for each feature\n", + " with tf.io.gfile.GFile(args.param_file, 'w') as f:\n", + " f.write(str(params))\n", + " return (x_train, y_train), (x_test, y_test)\n", + "\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_dnn_model():\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Dense(128, activation='relu', input_shape=(13,)),\n", + " tf.keras.layers.Dense(128, activation='relu'),\n", + " tf.keras.layers.Dense(1, activation='linear')\n", + " ])\n", + " model.compile(\n", + " loss='mse',\n", + " optimizer=tf.keras.optimizers.RMSprop(learning_rate=args.lr))\n", + " return model\n", + "\n", + "NUM_WORKERS = strategy.num_replicas_in_sync\n", + "# Here the batch size scales up by number of workers since\n", + "# `tf.data.Dataset.batch` expects the global batch size.\n", + "BATCH_SIZE = 16\n", + "GLOBAL_BATCH_SIZE = BATCH_SIZE * NUM_WORKERS\n", + "\n", + "with strategy.scope():\n", + " # Creation of dataset, and model building/compiling need to be within\n", + " # `strategy.scope()`.\n", + " model = build_and_compile_dnn_model()\n", + "\n", + "# Train the model\n", + "(x_train, y_train), (x_test, y_test) = make_dataset()\n", + "model.fit(x_train, y_train, epochs=args.epochs, batch_size=GLOBAL_BATCH_SIZE)\n", + "model.save(args.model_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tarball_training_script" + }, + "source": [ + "#### Store training script on your Cloud Storage bucket\n", + "\n", + "Next, you package the training folder into a compressed tar ball, and then store it in your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tarball_training_script" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_boston.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job" + }, + "source": [ + "### Train the model\n", + "\n", + "\n", + "Now start the training of your custom training job on Vertex. Use this helper function `create_custom_job`, which takes the following parameter:\n", + "\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "The helper function calls job client service's `create_custom_job` method, with the following parameters:\n", + "\n", + "-`parent`: The Vertex location path to `Dataset`, `Model` and `Endpoint` resources.\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "You will display a handful of the fields returned in `response` object, with the two that are of most interest are:\n", + "\n", + "`response.name`: The Vertex fully qualified identifier assigned to this custom training job. You save this identifier for using in subsequent steps.\n", + "\n", + "`response.state`: The current state of the custom training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job" + }, + "outputs": [], + "source": [ + "def create_custom_job(custom_job):\n", + " response = clients[\"job\"].create_custom_job(parent=PARENT, custom_job=custom_job)\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = create_custom_job(custom_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "job_id:response" + }, + "source": [ + "Now get the unique identifier for the custom job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom job\n", + "job_id = response.name\n", + "# The short numeric ID for the custom job\n", + "job_short_id = job_id.split(\"/\")[-1]\n", + "\n", + "print(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_custom_job" + }, + "source": [ + "### Get information on a custom job\n", + "\n", + "Next, use this helper function `get_custom_job`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "The helper function calls the job client service's`get_custom_job` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "If you recall, you got the Vertex fully qualified identifier for the custom job in the `response.name` field when you called the `create_custom_job` method, and saved the identifier in the variable `job_id`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_custom_job" + }, + "outputs": [], + "source": [ + "def get_custom_job(name, silent=False):\n", + " response = clients[\"job\"].get_custom_job(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = get_custom_job(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete:custom" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, we will need to know the location of the saved model, which the Python script saved in your local Cloud Storage bucket at `MODEL_DIR + '/saved_model.pb'`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete:custom" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_custom_job(job_id, True)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_path_to_deploy = None\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " if not DIRECT:\n", + " MODEL_DIR = MODEL_DIR + \"/model\"\n", + " model_path_to_deploy = MODEL_DIR\n", + " print(\"Training Time:\", response.update_time - response.create_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model_to_deploy:\", model_path_to_deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "load_saved_model" + }, + "source": [ + "## Load the saved model\n", + "\n", + "Your model is stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Now load it from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n", + "\n", + "To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "load_saved_model" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:tabular" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now let's find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load the Boston Housing test (holdout) data from `tf.keras.datasets`, using the method `load_data()`. This will return the dataset as a tuple of two elements. The first element is the training data and the second is the test data. Each element is also a tuple of two elements: the feature data, and the corresponding labels (median value of owner-occupied home).\n", + "\n", + "You don't need the training data, and hence why we loaded it as `(_, _)`.\n", + "\n", + "Before you can run the data through evaluation, you need to preprocess it:\n", + "\n", + "x_test:\n", + "1. Normalize (rescaling) the data in each column by dividing each value by the maximum value of that column. This will replace each single value with a 32-bit floating point number between 0 and 1." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "evaluate_custom_model:tabular,boston" + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "from tensorflow.keras.datasets import boston_housing\n", + "\n", + "(_, _), (x_test, y_test) = boston_housing.load_data(\n", + " path=\"boston_housing.npz\", test_split=0.2, seed=113\n", + ")\n", + "\n", + "\n", + "def scale(feature):\n", + " max = np.max(feature)\n", + " feature = (feature / max).astype(np.float32)\n", + " return feature\n", + "\n", + "\n", + "# Let's save one data item that has not been scaled\n", + "x_test_notscaled = x_test[0:1].copy()\n", + "\n", + "for _ in range(13):\n", + " x_test[_] = scale(x_test[_])\n", + "x_test = x_test.astype(np.float32)\n", + "\n", + "print(x_test.shape, x_test.dtype, y_test.shape)\n", + "print(\"scaled\", x_test[0])\n", + "print(\"unscaled\", x_test_notscaled)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "perform_evaluation_custom" + }, + "source": [ + "### Perform the model evaluation\n", + "\n", + "Now evaluate how well the model in the custom job did." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "perform_evaluation_custom" + }, + "outputs": [], + "source": [ + "model.evaluate(x_test, y_test)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "how_serving_function_works" + }, + "source": [ + "## Upload the model for serving\n", + "\n", + "Next, you will upload your TF.Keras model from the custom job to Vertex `Model` service, which will create a Vertex `Model` resource for your custom model. During upload, you need to define a serving function to convert data to the format your model expects. If you send encoded data to Vertex, your serving function ensures that the data is decoded on the model server before it is passed as input to your model.\n", + "\n", + "### How does the serving function work\n", + "\n", + "When you send a request to an online prediction server, the request is received by a HTTP server. The HTTP server extracts the prediction request from the HTTP request content body. The extracted prediction request is forwarded to the serving function. For Google pre-built prediction containers, the request content is passed to the serving function as a `tf.string`.\n", + "\n", + "The serving function consists of two parts:\n", + "\n", + "- `preprocessing function`:\n", + " - Converts the input (`tf.string`) to the input shape and data type of the underlying model (dynamic graph).\n", + " - Performs the same preprocessing of the data that was done during training the underlying model -- e.g., normalizing, scaling, etc.\n", + "- `post-processing function`:\n", + " - Converts the model output to format expected by the receiving application -- e.q., compresses the output.\n", + " - Packages the output for the the receiving application -- e.g., add headings, make JSON object, etc.\n", + "\n", + "Both the preprocessing and post-processing functions are converted to static graphs which are fused to the model. The output from the underlying model is passed to the post-processing function. The post-processing function passes the converted/packaged output back to the HTTP server. The HTTP server returns the output as the HTTP response content.\n", + "\n", + "One consideration you need to consider when building serving functions for TF.Keras models is that they run as static graphs. That means, you cannot use TF graph operations that require a dynamic graph. If you do, you will get an error during the compile of the serving function which will indicate that you are using an EagerTensor which is not supported." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_signature" + }, + "source": [ + "## Get the serving function signature\n", + "\n", + "You can get the signatures of your model's input and output layers by reloading the model into memory, and querying it for the signatures corresponding to each layer.\n", + "\n", + "When making a prediction request, you need to route the request to the serving function instead of the model, so you need to know the input layer name of the serving function -- which you will use later when you make a prediction request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_signature" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_path_to_deploy)\n", + "\n", + "serving_input = list(\n", + " loaded.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "print(\"Serving function input:\", serving_input)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "upload_the_model:tfserving" + }, + "source": [ + "### Upload the model\n", + "\n", + "Use this helper function `upload_model` to upload your model, stored in SavedModel format, up to the `Model` service, which will instantiate a Vertex `Model` resource instance for your model. Once you've done that, you can use the `Model` resource instance in the same way as any other Vertex `Model` resource instance, such as deploying to an `Endpoint` resource for serving predictions.\n", + "\n", + "The helper function takes the following parameters:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` service.\n", + "- `image_uri`: The container image for the model deployment.\n", + "- `model_uri`: The Cloud Storage path to our SavedModel artificat. For this tutorial, this is the Cloud Storage location where the `trainer/task.py` saved the model artifacts, which we specified in the variable `MODEL_DIR`.\n", + "\n", + "The helper function calls the `Model` client service's method `upload_model`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for `Dataset`, `Model` and `Endpoint` resources.\n", + "- `model`: The specification for the Vertex `Model` resource instance.\n", + "\n", + "Let's now dive deeper into the Vertex model specification `model`. This is a dictionary object that consists of the following fields:\n", + "\n", + "- `display_name`: A human readable name for the `Model` resource.\n", + "- `metadata_schema_uri`: Since your model was built without an Vertex `Dataset` resource, you will leave this blank (`''`).\n", + "- `artificat_uri`: The Cloud Storage path where the model is stored in SavedModel format.\n", + "- `container_spec`: This is the specification for the Docker container that will be installed on the endpoint, from which the model will serve predictions. In this tutoriak, you use a TF Serving as a custom prediction container. Since this is a custom container, you need to tell Vertex how to start up and communicate with the serving binary:\n", + " - `command`: The serving binary (HTTP Server) to start up.\n", + " - `args`: The arguments to pass to the serving binary. For TF Serving, the required arguments are:\n", + " - `--model_name`: The human readable name to assign to the model.\n", + " - `--model_base_name`: Where to store the model artifacts in the container. The Vertex service sets the variable $(AIP_STORAGE_URI) to where the service installed the model artifacts in the container.\n", + " - `--rest_api_port`: The port to which to send REST based prediction requests. Can either be 8080 or 8501 (default for TF Serving).\n", + " - `--port`: The port to which to send gRPC based prediction requests. Should be 8500 for TF Serving.\n", + "\n", + " - `health_route`: The URL for the service to periodically ping for a response to verify that the serving binary is running. For TF Serving, this will be /v1/models/[model_name].\n", + " - `predict_route`: The URL for the service to route REST-based prediction requests to. For TF Serving, this will be /v1/models/[model_name]:predict.\n", + "\n", + "Uploading a model into a Vertex Model resource returns a long running operation, since it may take a few moments. You call response.result(), which is a synchronous call and will return when the Vertex Model resource is ready.\n", + "\n", + "The helper function returns the Vertex fully qualified identifier for the corresponding Vertex Model instance upload_model_response.model. You will save the identifier for subsequent steps in the variable model_to_deploy_id.\n", + "\n", + "*Note*: You drop the ending number subfolder (e.g., /1) from the model path to upload. The Vertex service will upload the parent folder above the subfolder with the model artifacts -- which is what TensorFlow serving binary expects." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "upload_the_model:tfserving" + }, + "outputs": [], + "source": [ + "IMAGE_URI = DEPLOY_IMAGE\n", + "MODEL_NAME = \"boston-\" + TIMESTAMP\n", + "\n", + "\n", + "def upload_model(display_name, image_uri, model_uri):\n", + " model = {\n", + " \"display_name\": display_name,\n", + " \"metadata_schema_uri\": \"\",\n", + " \"artifact_uri\": model_uri,\n", + " \"container_spec\": {\n", + " \"image_uri\": image_uri,\n", + " \"command\": [\"/usr/bin/tensorflow_model_server\"],\n", + " \"args\": [\n", + " \"--model_name=\" + MODEL_NAME,\n", + " \"--model_base_path=\" + \"$(AIP_STORAGE_URI)\",\n", + " \"--rest_api_port=8080\",\n", + " \"--port=8500\",\n", + " \"--file_system_poll_wait_seconds=31540000\",\n", + " ],\n", + " \"health_route\": \"/v1/models/\" + MODEL_NAME,\n", + " \"predict_route\": \"/v1/models/\" + MODEL_NAME + \":predict\",\n", + " },\n", + " }\n", + " response = clients[\"model\"].upload_model(parent=PARENT, model=model)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " upload_model_response = response.result(timeout=180)\n", + " print(\"upload_model_response\")\n", + " print(\" model:\", upload_model_response.model)\n", + " return upload_model_response.model\n", + "\n", + "\n", + "model_to_deploy_id = upload_model(\n", + " \"boston-\" + TIMESTAMP, IMAGE_URI, model_path_to_deploy[:-2]\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_model" + }, + "source": [ + "### Get `Model` resource information\n", + "\n", + "Now let's get the model information for just your model. Use this helper function `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource.\n", + "\n", + "This helper function calls the Vertex `Model` client service's method `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_model" + }, + "outputs": [], + "source": [ + "def get_model(name):\n", + " response = clients[\"model\"].get_model(name=name)\n", + " print(response)\n", + "\n", + "\n", + "get_model(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:custom" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex custom `Model` resource. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"boston_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:dedicated" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `dedicated_resources`: This refers to how many compute instances (replicas) that are scaled for serving prediction requests.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `min_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:dedicated" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"boston_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " \"machine_spec\": machine_spec,\n", + " },\n", + " \"disable_container_logging\": False,\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_item:test,tabular" + }, + "source": [ + "### Get test item\n", + "\n", + "You will use an example out of the test (holdout) portion of the dataset as a test item." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:test,tabular" + }, + "outputs": [], + "source": [ + "test_item = x_test[0]\n", + "test_label = y_test[0]\n", + "print(test_item.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "send_prediction_request:tabular" + }, + "source": [ + "### Send the prediction request\n", + "\n", + "Ok, now you have a test data item. Use this helper function `predict_data`, which takes the parameters:\n", + "\n", + "- `data`: The test data item as a numpy 1D array of floating point values.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `parameters_dict`: Additional parameters for serving.\n", + "\n", + "This function uses the prediction client service and calls the `predict` method with the parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `instances`: A list of instances (data items) to predict.\n", + "- `parameters`: Additional parameters for serving.\n", + "\n", + "To pass the test data to the prediction service, you package it for transmission to the serving binary as follows:\n", + "\n", + " 1. Convert the data item from a 1D numpy array to a 1D Python list.\n", + " 2. Convert the prediction request to a serialized Google protobuf (`json_format.ParseDict()`)\n", + "\n", + "\n", + "Each instance in the prediction request is a dictionary entry of the form:\n", + "\n", + " {input_name: content}\n", + "\n", + "- `input_name`: the name of the input layer of the underlying model.\n", + "- `content`: The data item as a 1D Python list.\n", + "\n", + "Since the `predict()` service can take multiple data items (instances), you will send your single data item as a list of one data item. As a final step, you package the instances list into Google's protobuf format -- which is what we pass to the `predict()` service.\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding image in the request. You will see in the output for each prediction:\n", + "\n", + "- `predictions` -- the predicated median value of a house in units of 1K USD." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "send_prediction_request:tabular" + }, + "outputs": [], + "source": [ + "def predict_data(data, endpoint, parameters_dict):\n", + " parameters = json_format.ParseDict(parameters_dict, Value())\n", + "\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [{serving_input: data.tolist()}]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", prediction)\n", + "\n", + "\n", + "predict_data(test_item, endpoint_id, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_custom_tabular_regression_online_tfserving.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_custom_text_binary_classification_batch.ipynb b/notebooks/community/gapic/custom/showcase_custom_text_binary_classification_batch.ipynb new file mode 100644 index 000000000..7e37c551d --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_custom_text_binary_classification_batch.ipynb @@ -0,0 +1,2112 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Custom training text binary classification model for batch prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to train and deploy a custom text binary classification model for batch prediction." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:custom,imdb,tbn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [IMDB Movie Reviews](https://www.tensorflow.org/datasets/catalog/imdb_reviews) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts whether a review is positive or negative in sentiment." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,training,batch_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create a custom model, with a training pipeline, from a Python script in a Google prebuilt Docker container using the Vertex client library, and then do a batch prediction on the uploaded model. You can alternatively create custom models using `gcloud` command-line tool or online using Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex custom job for training a model.\n", + "- Train the TensorFlow model.\n", + "- Retrieve and load the model artifacts.\n", + "- View the model evaluation.\n", + "- Upload the model as a Vertex `Model` resource.\n", + "- Make a batch prediction." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:training,cpu,prediction,cpu" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for training and prediction.\n", + "\n", + "Set the variables `TRAIN_GPU/TRAIN_NGPU` and `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU.\n", + "\n", + "*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3 -- which is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:training,cpu,prediction,cpu" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n", + " TRAIN_GPU, TRAIN_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n", + " )\n", + "else:\n", + " TRAIN_GPU, TRAIN_NGPU = (None, None)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (None, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:training,prediction" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for training and prediction\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-3:latest`\n", + " - TensorFlow 2.4\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-4:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/training/xgboost-cpu.1-1`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/training/scikit-learn-cpu.0-23:latest`\n", + " - Pytorch\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-5:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-6:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-7:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for training](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers).\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-3:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-90:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-82:latest`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-23:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-22:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-20:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for prediction](https://cloud.google.com/vertex-ai/docs/predictions/pre-built-containers)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf2-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf2-cpu.{}\".format(TF)\n", + "else:\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "TRAIN_IMAGE = \"gcr.io/cloud-aiplatform/training/{}:latest\".format(TRAIN_VERSION)\n", + "DEPLOY_IMAGE = \"gcr.io/cloud-aiplatform/prediction/{}:latest\".format(DEPLOY_VERSION)\n", + "\n", + "print(\"Training:\", TRAIN_IMAGE, TRAIN_GPU, TRAIN_NGPU)\n", + "print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU, DEPLOY_NGPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:training,prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for training and prediction.\n", + "\n", + "- Set the variables `TRAIN_COMPUTE` and `DEPLOY_COMPUTE` to configure the compute resources for the VMs you will use for for training and prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: The following is not supported for training:*\n", + "\n", + " - `standard`: 2 vCPUs\n", + " - `highcpu`: 2, 4 and 8 vCPUs\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_TRAIN_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Train machine type\", TRAIN_COMPUTE)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own custom model and training for IMDB Movie Reviews." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:custom" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Model Service for `Model` resources.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:custom" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"job\"] = create_job_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_model" + }, + "source": [ + "## Train a model\n", + "\n", + "There are two ways you can train a custom model using a container image:\n", + "\n", + "- **Use a Google Cloud prebuilt container**. If you use a prebuilt container, you will additionally specify a Python package to install into the container image. This Python package contains your code for training a custom model.\n", + "\n", + "- **Use your own custom container image**. If you use your own container, the container needs to contain your code for training a custom model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_specification:prebuilt_container" + }, + "source": [ + "## Prepare your custom job specification\n", + "\n", + "Now that your clients are ready, your first step is to create a Job Specification for your custom training job. The job specification will consist of the following:\n", + "\n", + "- `worker_pool_spec` : The specification of the type of machine(s) you will use for training and how many (single or distributed)\n", + "- `python_package_spec` : The specification of the Python package to be installed with the pre-built container." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "source": [ + "### Prepare your machine specification\n", + "\n", + "Now define the machine specification for your custom training job. This tells Vertex what type of machine instance to provision for the training.\n", + " - `machine_type`: The type of GCP instance to provision -- e.g., n1-standard-8.\n", + " - `accelerator_type`: The type, if any, of hardware accelerator. In this tutorial if you previously set the variable `TRAIN_GPU != None`, you are using a GPU; otherwise you will use a CPU.\n", + " - `accelerator_count`: The number of accelerators." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "outputs": [], + "source": [ + "if TRAIN_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": TRAIN_COMPUTE,\n", + " \"accelerator_type\": TRAIN_GPU,\n", + " \"accelerator_count\": TRAIN_NGPU,\n", + " }\n", + "else:\n", + " machine_spec = {\"machine_type\": TRAIN_COMPUTE, \"accelerator_count\": 0}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "source": [ + "### Prepare your disk specification\n", + "\n", + "(optional) Now define the disk specification for your custom training job. This tells Vertex what type and size of disk to provision in each machine instance for the training.\n", + "\n", + " - `boot_disk_type`: Either SSD or Standard. SSD is faster, and Standard is less expensive. Defaults to SSD.\n", + " - `boot_disk_size_gb`: Size of disk in GB." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "outputs": [], + "source": [ + "DISK_TYPE = \"pd-ssd\" # [ pd-ssd, pd-standard]\n", + "DISK_SIZE = 200 # GB\n", + "\n", + "disk_spec = {\"boot_disk_type\": DISK_TYPE, \"boot_disk_size_gb\": DISK_SIZE}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container" + }, + "source": [ + "### Define the worker pool specification\n", + "\n", + "Next, you define the worker pool specification for your custom training job. The worker pool specification will consist of the following:\n", + "\n", + "- `replica_count`: The number of instances to provision of this machine type.\n", + "- `machine_spec`: The hardware specification.\n", + "- `disk_spec` : (optional) The disk storage specification.\n", + "\n", + "- `python_package`: The Python training package to install on the VM instance(s) and which Python module to invoke, along with command line arguments for the Python module.\n", + "\n", + "Let's dive deeper now into the python package specification:\n", + "\n", + "-`executor_image_spec`: This is the docker image which is configured for your custom training job.\n", + "\n", + "-`package_uris`: This is a list of the locations (URIs) of your python training packages to install on the provisioned instance. The locations need to be in a Cloud Storage bucket. These can be either individual python files or a zip (archive) of an entire package. In the later case, the job service will unzip (unarchive) the contents into the docker image.\n", + "\n", + "-`python_module`: The Python module (script) to invoke for running the custom training job. In this example, you will be invoking `trainer.task.py` -- note that it was not neccessary to append the `.py` suffix.\n", + "\n", + "-`args`: The command line arguments to pass to the corresponding Pythom module. In this example, you will be setting:\n", + " - `\"--model-dir=\" + MODEL_DIR` : The Cloud Storage location where to store the model artifacts. There are two ways to tell the training script where to save the model artifacts:\n", + " - direct: You pass the Cloud Storage location as a command line argument to your training script (set variable `DIRECT = True`), or\n", + " - indirect: The service passes the Cloud Storage location as the environment variable `AIP_MODEL_DIR` to your training script (set variable `DIRECT = False`). In this case, you tell the service the model artifact location in the job specification.\n", + " - `\"--epochs=\" + EPOCHS`: The number of epochs for training.\n", + " - `\"--steps=\" + STEPS`: The number of steps (batches) per epoch.\n", + " - `\"--distribute=\" + TRAIN_STRATEGY\"` : The training distribution strategy to use for single or distributed training.\n", + " - `\"single\"`: single device.\n", + " - `\"mirror\"`: all GPU devices on a single compute instance.\n", + " - `\"multi\"`: all GPU devices on all compute instances." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_job_\" + TIMESTAMP\n", + "MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, JOB_NAME)\n", + "\n", + "if not TRAIN_NGPU or TRAIN_NGPU < 2:\n", + " TRAIN_STRATEGY = \"single\"\n", + "else:\n", + " TRAIN_STRATEGY = \"mirror\"\n", + "\n", + "EPOCHS = 20\n", + "STEPS = 100\n", + "\n", + "DIRECT = True\n", + "if DIRECT:\n", + " CMDARGS = [\n", + " \"--model-dir=\" + MODEL_DIR,\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "else:\n", + " CMDARGS = [\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "\n", + "worker_pool_spec = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": machine_spec,\n", + " \"disk_spec\": disk_spec,\n", + " \"python_package_spec\": {\n", + " \"executor_image_uri\": TRAIN_IMAGE,\n", + " \"package_uris\": [BUCKET_NAME + \"/trainer_imdb.tar.gz\"],\n", + " \"python_module\": \"trainer.task\",\n", + " \"args\": CMDARGS,\n", + " },\n", + " }\n", + "]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "assemble_custom_job_specification" + }, + "source": [ + "### Assemble a job specification\n", + "\n", + "Now assemble the complete description for the custom job specification:\n", + "\n", + "- `display_name`: The human readable name you assign to this custom job.\n", + "- `job_spec`: The specification for the custom job.\n", + " - `worker_pool_specs`: The specification for the machine VM instances.\n", + " - `base_output_directory`: This tells the service the Cloud Storage location where to save the model artifacts (when variable `DIRECT = False`). The service will then pass the location to the training script as the environment variable `AIP_MODEL_DIR`, and the path will be of the form:\n", + "\n", + " /model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "assemble_custom_job_specification" + }, + "outputs": [], + "source": [ + "if DIRECT:\n", + " job_spec = {\"worker_pool_specs\": worker_pool_spec}\n", + "else:\n", + " job_spec = {\n", + " \"worker_pool_specs\": worker_pool_spec,\n", + " \"base_output_directory\": {\"output_uri_prefix\": MODEL_DIR},\n", + " }\n", + "\n", + "custom_job = {\"display_name\": JOB_NAME, \"job_spec\": job_spec}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "examine_training_package" + }, + "source": [ + "### Examine the training package\n", + "\n", + "#### Package layout\n", + "\n", + "Before you start the training, you will look at how a Python package is assembled for a custom training job. When unarchived, the package contains the following directory/file layout.\n", + "\n", + "- PKG-INFO\n", + "- README.md\n", + "- setup.cfg\n", + "- setup.py\n", + "- trainer\n", + " - \\_\\_init\\_\\_.py\n", + " - task.py\n", + "\n", + "The files `setup.cfg` and `setup.py` are the instructions for installing the package into the operating environment of the Docker image.\n", + "\n", + "The file `trainer/task.py` is the Python script for executing the custom training job. *Note*, when we referred to it in the worker pool specification, we replace the directory slash with a dot (`trainer.task`) and dropped the file suffix (`.py`).\n", + "\n", + "#### Package Assembly\n", + "\n", + "In the following cells, you will assemble the training package." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "examine_training_package" + }, + "outputs": [], + "source": [ + "# Make folder for Python training script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\ntag_build =\\n\\ntag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'tensorflow_datasets==1.3.0',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\nName: IMDB Movie Reviews text binary classification\\n\\nVersion: 0.0.0\\n\\nSummary: Demostration training script\\n\\nHome-page: www.google.com\\n\\nAuthor: Google\\n\\nAuthor-email: aferlitsch@google.com\\n\\nLicense: Public\\n\\nDescription: Demo\\n\\nPlatform: Vertex\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:imdb" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the training script task.py. I won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Gets the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n", + "- Loads IMDB Movie Reviews dataset from TF Datasets (tfds).\n", + "- Builds a simple RNN model using TF.Keras model API.\n", + "- Compiles the model (`compile()`).\n", + "- Sets a training distribution strategy according to the argument `args.distribute`.\n", + "- Trains the model (`fit()`) with epochs specified by `args.epochs`.\n", + "- Saves the trained model (`save(args.model_dir)`) to the specified model directory." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:imdb" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Single, Mirror and Multi-Machine Distributed Training for IMDB\n", + "\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "import argparse\n", + "import os\n", + "import sys\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv('AIP_MODEL_DIR'), type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=1e-4, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=20, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=100, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "print(device_lib.list_local_devices())\n", + "\n", + "# Single Machine, single compute device\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "# Single Machine, multiple compute device\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "# Multiple Machine, multiple compute device\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "# Multi-worker configuration\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "# Preparing dataset\n", + "BUFFER_SIZE = 10000\n", + "BATCH_SIZE = 64\n", + "\n", + "\n", + "def make_datasets():\n", + " dataset, info = tfds.load('imdb_reviews/subwords8k', with_info=True,\n", + " as_supervised=True)\n", + " train_dataset, test_dataset = dataset['train'], dataset['test']\n", + " encoder = info.features['text'].encoder\n", + "\n", + " padded_shapes = ([None],())\n", + " return train_dataset.shuffle(BUFFER_SIZE).padded_batch(BATCH_SIZE, padded_shapes), encoder\n", + "\n", + "\n", + "train_dataset, encoder = make_datasets()\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_rnn_model(encoder):\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Embedding(encoder.vocab_size, 64),\n", + " tf.keras.layers.Bidirectional(tf.keras.layers.LSTM(64)),\n", + " tf.keras.layers.Dense(64, activation='relu'),\n", + " tf.keras.layers.Dense(1, activation='sigmoid')\n", + " ])\n", + " model.compile(loss=tf.keras.losses.BinaryCrossentropy(from_logits=True),\n", + " optimizer=tf.keras.optimizers.Adam(args.lr),\n", + " metrics=['accuracy'])\n", + " return model\n", + "\n", + "\n", + "with strategy.scope():\n", + " # Creation of dataset, and model building/compiling need to be within\n", + " # `strategy.scope()`.\n", + " model = build_and_compile_rnn_model(encoder)\n", + "\n", + "# Train the model\n", + "model.fit(train_dataset, epochs=args.epochs, steps_per_epoch=args.steps)\n", + "model.save(args.model_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tarball_training_script" + }, + "source": [ + "#### Store training script on your Cloud Storage bucket\n", + "\n", + "Next, you package the training folder into a compressed tar ball, and then store it in your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tarball_training_script" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_imdb.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job" + }, + "source": [ + "### Train the model\n", + "\n", + "\n", + "Now start the training of your custom training job on Vertex. Use this helper function `create_custom_job`, which takes the following parameter:\n", + "\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "The helper function calls job client service's `create_custom_job` method, with the following parameters:\n", + "\n", + "-`parent`: The Vertex location path to `Dataset`, `Model` and `Endpoint` resources.\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "You will display a handful of the fields returned in `response` object, with the two that are of most interest are:\n", + "\n", + "`response.name`: The Vertex fully qualified identifier assigned to this custom training job. You save this identifier for using in subsequent steps.\n", + "\n", + "`response.state`: The current state of the custom training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job" + }, + "outputs": [], + "source": [ + "def create_custom_job(custom_job):\n", + " response = clients[\"job\"].create_custom_job(parent=PARENT, custom_job=custom_job)\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = create_custom_job(custom_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "job_id:response" + }, + "source": [ + "Now get the unique identifier for the custom job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom job\n", + "job_id = response.name\n", + "# The short numeric ID for the custom job\n", + "job_short_id = job_id.split(\"/\")[-1]\n", + "\n", + "print(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_custom_job" + }, + "source": [ + "### Get information on a custom job\n", + "\n", + "Next, use this helper function `get_custom_job`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "The helper function calls the job client service's`get_custom_job` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "If you recall, you got the Vertex fully qualified identifier for the custom job in the `response.name` field when you called the `create_custom_job` method, and saved the identifier in the variable `job_id`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_custom_job" + }, + "outputs": [], + "source": [ + "def get_custom_job(name, silent=False):\n", + " response = clients[\"job\"].get_custom_job(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = get_custom_job(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete:custom" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, we will need to know the location of the saved model, which the Python script saved in your local Cloud Storage bucket at `MODEL_DIR + '/saved_model.pb'`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete:custom" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_custom_job(job_id, True)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_path_to_deploy = None\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " if not DIRECT:\n", + " MODEL_DIR = MODEL_DIR + \"/model\"\n", + " model_path_to_deploy = MODEL_DIR\n", + " print(\"Training Time:\", response.update_time - response.create_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model_to_deploy:\", model_path_to_deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "load_saved_model" + }, + "source": [ + "## Load the saved model\n", + "\n", + "Your model is stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Now load it from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n", + "\n", + "To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "load_saved_model" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:text" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now let's find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load the IMDB Movie Review test (holdout) data from `tfds.datasets`, using the method `load()`. This will return the dataset as a tuple of two elements. The first element is the dataset and the second is information on the dataset, which will contain the predefined vocabulary encoder. The encoder will convert words into a numerical embedding, which was pretrained and used in the custom training script.\n", + "\n", + "\n", + "When you trained the model, you needed to set a fix input length for your text. For forward feeding batches, the `padded_batch()` property of the corresponding `tf.dataset` was set to pad each input sequence into the same shape for a batch.\n", + "\n", + "For the test data, you also need to set the `padded_batch()` property accordingly." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "evaluate_custom_model:text,imdb" + }, + "outputs": [], + "source": [ + "import tensorflow_datasets as tfds\n", + "\n", + "dataset, info = tfds.load(\"imdb_reviews/subwords8k\", with_info=True, as_supervised=True)\n", + "test_dataset = dataset[\"test\"]\n", + "encoder = info.features[\"text\"].encoder\n", + "\n", + "BATCH_SIZE = 64\n", + "padded_shapes = ([None], ())\n", + "test_dataset = test_dataset.padded_batch(BATCH_SIZE, padded_shapes)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "perform_evaluation_custom:tfds" + }, + "source": [ + "### Perform the model evaluation\n", + "\n", + "Now evaluate how well the model in the custom job did." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "perform_evaluation_custom:tfds" + }, + "outputs": [], + "source": [ + "model.evaluate(test_dataset)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "how_serving_function_works" + }, + "source": [ + "## Upload the model for serving\n", + "\n", + "Next, you will upload your TF.Keras model from the custom job to Vertex `Model` service, which will create a Vertex `Model` resource for your custom model. During upload, you need to define a serving function to convert data to the format your model expects. If you send encoded data to Vertex, your serving function ensures that the data is decoded on the model server before it is passed as input to your model.\n", + "\n", + "### How does the serving function work\n", + "\n", + "When you send a request to an online prediction server, the request is received by a HTTP server. The HTTP server extracts the prediction request from the HTTP request content body. The extracted prediction request is forwarded to the serving function. For Google pre-built prediction containers, the request content is passed to the serving function as a `tf.string`.\n", + "\n", + "The serving function consists of two parts:\n", + "\n", + "- `preprocessing function`:\n", + " - Converts the input (`tf.string`) to the input shape and data type of the underlying model (dynamic graph).\n", + " - Performs the same preprocessing of the data that was done during training the underlying model -- e.g., normalizing, scaling, etc.\n", + "- `post-processing function`:\n", + " - Converts the model output to format expected by the receiving application -- e.q., compresses the output.\n", + " - Packages the output for the the receiving application -- e.g., add headings, make JSON object, etc.\n", + "\n", + "Both the preprocessing and post-processing functions are converted to static graphs which are fused to the model. The output from the underlying model is passed to the post-processing function. The post-processing function passes the converted/packaged output back to the HTTP server. The HTTP server returns the output as the HTTP response content.\n", + "\n", + "One consideration you need to consider when building serving functions for TF.Keras models is that they run as static graphs. That means, you cannot use TF graph operations that require a dynamic graph. If you do, you will get an error during the compile of the serving function which will indicate that you are using an EagerTensor which is not supported." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_signature" + }, + "source": [ + "## Get the serving function signature\n", + "\n", + "You can get the signatures of your model's input and output layers by reloading the model into memory, and querying it for the signatures corresponding to each layer.\n", + "\n", + "When making a prediction request, you need to route the request to the serving function instead of the model, so you need to know the input layer name of the serving function -- which you will use later when you make a prediction request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_signature" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_path_to_deploy)\n", + "\n", + "serving_input = list(\n", + " loaded.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "print(\"Serving function input:\", serving_input)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "upload_the_model" + }, + "source": [ + "### Upload the model\n", + "\n", + "Use this helper function `upload_model` to upload your model, stored in SavedModel format, up to the `Model` service, which will instantiate a Vertex `Model` resource instance for your model. Once you've done that, you can use the `Model` resource instance in the same way as any other Vertex `Model` resource instance, such as deploying to an `Endpoint` resource for serving predictions.\n", + "\n", + "The helper function takes the following parameters:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` service.\n", + "- `image_uri`: The container image for the model deployment.\n", + "- `model_uri`: The Cloud Storage path to our SavedModel artificat. For this tutorial, this is the Cloud Storage location where the `trainer/task.py` saved the model artifacts, which we specified in the variable `MODEL_DIR`.\n", + "\n", + "The helper function calls the `Model` client service's method `upload_model`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for `Dataset`, `Model` and `Endpoint` resources.\n", + "- `model`: The specification for the Vertex `Model` resource instance.\n", + "\n", + "Let's now dive deeper into the Vertex model specification `model`. This is a dictionary object that consists of the following fields:\n", + "\n", + "- `display_name`: A human readable name for the `Model` resource.\n", + "- `metadata_schema_uri`: Since your model was built without an Vertex `Dataset` resource, you will leave this blank (`''`).\n", + "- `artificat_uri`: The Cloud Storage path where the model is stored in SavedModel format.\n", + "- `container_spec`: This is the specification for the Docker container that will be installed on the `Endpoint` resource, from which the `Model` resource will serve predictions. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + "\n", + "Uploading a model into a Vertex Model resource returns a long running operation, since it may take a few moments. You call response.result(), which is a synchronous call and will return when the Vertex Model resource is ready.\n", + "\n", + "The helper function returns the Vertex fully qualified identifier for the corresponding Vertex Model instance upload_model_response.model. You will save the identifier for subsequent steps in the variable model_to_deploy_id." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "upload_the_model" + }, + "outputs": [], + "source": [ + "IMAGE_URI = DEPLOY_IMAGE\n", + "\n", + "\n", + "def upload_model(display_name, image_uri, model_uri):\n", + " model = {\n", + " \"display_name\": display_name,\n", + " \"metadata_schema_uri\": \"\",\n", + " \"artifact_uri\": model_uri,\n", + " \"container_spec\": {\n", + " \"image_uri\": image_uri,\n", + " \"command\": [],\n", + " \"args\": [],\n", + " \"env\": [{\"name\": \"env_name\", \"value\": \"env_value\"}],\n", + " \"ports\": [{\"container_port\": 8080}],\n", + " \"predict_route\": \"\",\n", + " \"health_route\": \"\",\n", + " },\n", + " }\n", + " response = clients[\"model\"].upload_model(parent=PARENT, model=model)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " upload_model_response = response.result(timeout=180)\n", + " print(\"upload_model_response\")\n", + " print(\" model:\", upload_model_response.model)\n", + " return upload_model_response.model\n", + "\n", + "\n", + "model_to_deploy_id = upload_model(\"imdb-\" + TIMESTAMP, IMAGE_URI, model_path_to_deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_model" + }, + "source": [ + "### Get `Model` resource information\n", + "\n", + "Now let's get the model information for just your model. Use this helper function `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource.\n", + "\n", + "This helper function calls the Vertex `Model` client service's method `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_model" + }, + "outputs": [], + "source": [ + "def get_model(name):\n", + " response = clients[\"model\"].get_model(name=name)\n", + " print(response)\n", + "\n", + "\n", + "get_model(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy:batch_prediction" + }, + "source": [ + "## Model deployment for batch prediction\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created for batch prediction. This differs from deploying a `Model` resource for on-demand prediction.\n", + "\n", + "For online prediction, you:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource.\n", + "\n", + "3. Make online prediction requests to the `Endpoint` resource.\n", + "\n", + "For batch-prediction, you:\n", + "\n", + "1. Create a batch prediction job.\n", + "\n", + "2. The job service will provision resources for the batch prediction request.\n", + "\n", + "3. The results of the batch prediction request are returned to the caller.\n", + "\n", + "4. The job service will unprovision the resoures for the batch prediction request." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a batch prediction request\n", + "\n", + "Now do a batch prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "prepare_test_item:text" + }, + "source": [ + "### Prepare the request content\n", + "\n", + "Since the dataset is a `tf.dataset`, which acts as a generator, we must use it as an iterator to access the data items in the test data. We do the following to get a single data item from the test data:\n", + "\n", + "- Set the property for the number of batches to draw per iteration to one using the method `take(1)`.\n", + "- Iterate once through the test data -- i.e., we do a break within the for loop.\n", + "- In the single iteration, we save the data item which is in the form of a tuple.\n", + "- The data item will be the first element of the tuple, which you then will convert from an tensor to a numpy array -- `data[0].numpy()`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "prepare_test_item:text" + }, + "outputs": [], + "source": [ + "import tensorflow_datasets as tfds\n", + "\n", + "dataset, info = tfds.load(\"imdb_reviews/subwords8k\", with_info=True, as_supervised=True)\n", + "test_dataset = dataset[\"test\"]\n", + "\n", + "test_dataset.take(1)\n", + "for data in test_dataset:\n", + " print(data)\n", + " break\n", + "\n", + "test_item = data[0].numpy()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:custom,text" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "Now make a batch input file, which you will store in your local Cloud Storage bucket. Each instance in the prediction request is a dictionary entry of the form:\n", + "\n", + " {serving_input: content}\n", + "\n", + "- `input_name`: the name of the input layer of the underlying model.\n", + "- `content`: The text data item encoded as an embedding." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_file:custom,text" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "gcs_input_uri = BUCKET_NAME + \"/\" + \"test.jsonl\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " data = {serving_input: test_item.tolist()}\n", + " f.write(json.dumps(data) + \"\\n\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your batch prediction requests:\n", + "\n", + "- Single Instance: The batch prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The batch prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and batch prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The batch prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_request:custom" + }, + "source": [ + "### Make batch prediction request\n", + "\n", + "Now that your batch of two test items is ready, let's do the batch request. Use this helper function `create_batch_prediction_job`, with the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the prediction job.\n", + "- `model_name`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `gcs_source_uri`: The Cloud Storage path to the input file -- which you created above.\n", + "- `gcs_destination_output_uri_prefix`: The Cloud Storage path that the service will write the predictions to.\n", + "- `parameters`: Additional filtering parameters for serving prediction results.\n", + "\n", + "The helper function calls the job client service's `create_batch_prediction_job` metho, with the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for Dataset, Model and Pipeline resources.\n", + "- `batch_prediction_job`: The specification for the batch prediction job.\n", + "\n", + "Let's now dive into the specification for the `batch_prediction_job`:\n", + "\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the `Model` resource.\n", + "- `dedicated_resources`: The compute resources to provision for the batch prediction job.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `starting_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "- `model_parameters`: Additional filtering parameters for serving prediction results. No Additional parameters are supported for custom models.\n", + "- `input_config`: The input source and format type for the instances to predict.\n", + " - `instances_format`: The format of the batch prediction request file: `csv` or `jsonl`.\n", + " - `gcs_source`: A list of one or more Cloud Storage paths to your batch prediction requests.\n", + "- `output_config`: The output destination and format for the predictions.\n", + " - `prediction_format`: The format of the batch prediction response file: `csv` or `jsonl`.\n", + " - `gcs_destination`: The output destination for the predictions.\n", + "\n", + "This call is an asychronous operation. You will print from the response object a few select fields, including:\n", + "\n", + "- `name`: The Vertex fully qualified identifier assigned to the batch prediction job.\n", + "- `display_name`: The human readable name for the prediction batch job.\n", + "- `model`: The Vertex fully qualified identifier for the Model resource.\n", + "- `generate_explanations`: Whether True/False explanations were provided with the predictions (explainability).\n", + "- `state`: The state of the prediction job (pending, running, etc).\n", + "\n", + "Since this call will take a few moments to execute, you will likely get `JobState.JOB_STATE_PENDING` for `state`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_request:custom" + }, + "outputs": [], + "source": [ + "BATCH_MODEL = \"imdb_batch-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_batch_prediction_job(\n", + " display_name,\n", + " model_name,\n", + " gcs_source_uri,\n", + " gcs_destination_output_uri_prefix,\n", + " parameters=None,\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " batch_prediction_job = {\n", + " \"display_name\": display_name,\n", + " # Format: 'projects/{project}/locations/{location}/models/{model_id}'\n", + " \"model\": model_name,\n", + " \"model_parameters\": json_format.ParseDict(parameters, Value()),\n", + " \"input_config\": {\n", + " \"instances_format\": IN_FORMAT,\n", + " \"gcs_source\": {\"uris\": [gcs_source_uri]},\n", + " },\n", + " \"output_config\": {\n", + " \"predictions_format\": OUT_FORMAT,\n", + " \"gcs_destination\": {\"output_uri_prefix\": gcs_destination_output_uri_prefix},\n", + " },\n", + " \"dedicated_resources\": {\n", + " \"machine_spec\": machine_spec,\n", + " \"starting_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " },\n", + " }\n", + " response = clients[\"job\"].create_batch_prediction_job(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + " )\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try:\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", response.labels)\n", + " return response\n", + "\n", + "\n", + "IN_FORMAT = \"jsonl\"\n", + "OUT_FORMAT = \"jsonl\"\n", + "\n", + "response = create_batch_prediction_job(\n", + " BATCH_MODEL, model_to_deploy_id, gcs_input_uri, BUCKET_NAME\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batch_job_id:response" + }, + "source": [ + "Now get the unique identifier for the batch prediction job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batch_job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the batch job\n", + "batch_job_id = response.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split(\"/\")[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_batch_prediction_job" + }, + "source": [ + "### Get information on a batch prediction job\n", + "\n", + "Use this helper function `get_batch_prediction_job`, with the following paramter:\n", + "\n", + "- `job_name`: The Vertex fully qualified identifier for the batch prediction job.\n", + "\n", + "The helper function calls the job client service's `get_batch_prediction_job` method, with the following paramter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the batch prediction job. In this tutorial, you will pass it the Vertex fully qualified identifier for your batch prediction job -- `batch_job_id`\n", + "\n", + "The helper function will return the Cloud Storage path to where the predictions are stored -- `gcs_destination`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_batch_prediction_job" + }, + "outputs": [], + "source": [ + "def get_batch_prediction_job(job_name, silent=False):\n", + " response = clients[\"job\"].get_batch_prediction_job(name=job_name)\n", + " if silent:\n", + " return response.output_config.gcs_destination.output_uri_prefix, response.state\n", + "\n", + " print(\"response\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" model:\", response.model)\n", + " try: # not all data types support explanations\n", + " print(\" generate_explanation:\", response.generate_explanation)\n", + " except:\n", + " pass\n", + " print(\" state:\", response.state)\n", + " print(\" error:\", response.error)\n", + " gcs_destination = response.output_config.gcs_destination\n", + " print(\" gcs_destination\")\n", + " print(\" output_uri_prefix:\", gcs_destination.output_uri_prefix)\n", + " return gcs_destination.output_uri_prefix, response.state\n", + "\n", + "\n", + "predictions, state = get_batch_prediction_job(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_the_predictions:custom,tbn" + }, + "source": [ + "### Get the predictions\n", + "\n", + "When the batch prediction is done processing, the job state will be `JOB_STATE_SUCCEEDED`.\n", + "\n", + "Finally you view the predictions stored at the Cloud Storage path you set as output. The predictions will be in a JSONL format, which you indicated at the time you made the batch prediction job, under a subfolder starting with the name `prediction`, and under that folder will be a file called `prediction.results-xxxxx-of-xxxxx`.\n", + "\n", + "Now display (cat) the contents. You will see multiple JSON objects, one for each prediction.\n", + "\n", + "The response contains a JSON object for each instance, in the form:\n", + "\n", + "- `embedding_input`: The input for the prediction.\n", + "- `predictions` -- the predicated binary sentiment between 0 (negative) and 1 (positive)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_the_predictions:custom,text" + }, + "outputs": [], + "source": [ + "def get_latest_predictions(gcs_out_dir):\n", + " \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n", + " folders = !gsutil ls $gcs_out_dir\n", + " latest = \"\"\n", + " for folder in folders:\n", + " subfolder = folder.split(\"/\")[-2]\n", + " if subfolder.startswith(\"prediction-\"):\n", + " if subfolder > latest:\n", + " latest = folder[:-1]\n", + " return latest\n", + "\n", + "\n", + "while True:\n", + " predictions, state = get_batch_prediction_job(batch_job_id, True)\n", + " if state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", state)\n", + " if state == aip.JobState.JOB_STATE_FAILED:\n", + " raise Exception(\"Batch Job Failed\")\n", + " else:\n", + " folder = get_latest_predictions(predictions)\n", + " ! gsutil ls $folder/prediction.results*\n", + "\n", + " print(\"Results:\")\n", + " ! gsutil cat $folder/prediction.results*\n", + "\n", + " print(\"Errors:\")\n", + " ! gsutil cat $folder/prediction.errors*\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_custom_text_binary_classification_batch.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_custom_text_binary_classification_online.ipynb b/notebooks/community/gapic/custom/showcase_custom_text_binary_classification_online.ipynb new file mode 100644 index 000000000..73b3bd948 --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_custom_text_binary_classification_online.ipynb @@ -0,0 +1,2090 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Custom training text binary classification model for online prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to train and deploy a custom text binary classification model for online prediction." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:custom,imdb,tbn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [IMDB Movie Reviews](https://www.tensorflow.org/datasets/catalog/imdb_reviews) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts whether a review is positive or negative in sentiment." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,training,online_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create a custom model from a Python script in a Google prebuilt Docker container using the Vertex client library, and then do a prediction on the deployed model by sending data. You can alternatively create custom models using `gcloud` command-line tool or online using Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex custom job for training a model.\n", + "- Train a TensorFlow model.\n", + "- Retrieve and load the model artifacts.\n", + "- View the model evaluation.\n", + "- Upload the model as a Vertex `Model` resource.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:training,cpu,prediction,cpu" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for training and prediction.\n", + "\n", + "Set the variables `TRAIN_GPU/TRAIN_NGPU` and `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU.\n", + "\n", + "*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3 -- which is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:training,cpu,prediction,cpu" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n", + " TRAIN_GPU, TRAIN_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n", + " )\n", + "else:\n", + " TRAIN_GPU, TRAIN_NGPU = (None, None)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (None, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:training,prediction" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for training and prediction\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-3:latest`\n", + " - TensorFlow 2.4\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-4:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/training/xgboost-cpu.1-1`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/training/scikit-learn-cpu.0-23:latest`\n", + " - Pytorch\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-5:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-6:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-7:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for training](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers).\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-3:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-90:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-82:latest`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-23:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-22:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-20:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for prediction](https://cloud.google.com/vertex-ai/docs/predictions/pre-built-containers)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf2-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf2-cpu.{}\".format(TF)\n", + "else:\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "TRAIN_IMAGE = \"gcr.io/cloud-aiplatform/training/{}:latest\".format(TRAIN_VERSION)\n", + "DEPLOY_IMAGE = \"gcr.io/cloud-aiplatform/prediction/{}:latest\".format(DEPLOY_VERSION)\n", + "\n", + "print(\"Training:\", TRAIN_IMAGE, TRAIN_GPU, TRAIN_NGPU)\n", + "print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU, DEPLOY_NGPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:training,prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for training and prediction.\n", + "\n", + "- Set the variables `TRAIN_COMPUTE` and `DEPLOY_COMPUTE` to configure the compute resources for the VMs you will use for for training and prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: The following is not supported for training:*\n", + "\n", + " - `standard`: 2 vCPUs\n", + " - `highcpu`: 2, 4 and 8 vCPUs\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_TRAIN_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Train machine type\", TRAIN_COMPUTE)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own custom model and training for IMDB Movie Reviews." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:custom" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Model Service for `Model` resources.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:custom" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"job\"] = create_job_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_model" + }, + "source": [ + "## Train a model\n", + "\n", + "There are two ways you can train a custom model using a container image:\n", + "\n", + "- **Use a Google Cloud prebuilt container**. If you use a prebuilt container, you will additionally specify a Python package to install into the container image. This Python package contains your code for training a custom model.\n", + "\n", + "- **Use your own custom container image**. If you use your own container, the container needs to contain your code for training a custom model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_specification:prebuilt_container" + }, + "source": [ + "## Prepare your custom job specification\n", + "\n", + "Now that your clients are ready, your first step is to create a Job Specification for your custom training job. The job specification will consist of the following:\n", + "\n", + "- `worker_pool_spec` : The specification of the type of machine(s) you will use for training and how many (single or distributed)\n", + "- `python_package_spec` : The specification of the Python package to be installed with the pre-built container." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "source": [ + "### Prepare your machine specification\n", + "\n", + "Now define the machine specification for your custom training job. This tells Vertex what type of machine instance to provision for the training.\n", + " - `machine_type`: The type of GCP instance to provision -- e.g., n1-standard-8.\n", + " - `accelerator_type`: The type, if any, of hardware accelerator. In this tutorial if you previously set the variable `TRAIN_GPU != None`, you are using a GPU; otherwise you will use a CPU.\n", + " - `accelerator_count`: The number of accelerators." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "outputs": [], + "source": [ + "if TRAIN_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": TRAIN_COMPUTE,\n", + " \"accelerator_type\": TRAIN_GPU,\n", + " \"accelerator_count\": TRAIN_NGPU,\n", + " }\n", + "else:\n", + " machine_spec = {\"machine_type\": TRAIN_COMPUTE, \"accelerator_count\": 0}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "source": [ + "### Prepare your disk specification\n", + "\n", + "(optional) Now define the disk specification for your custom training job. This tells Vertex what type and size of disk to provision in each machine instance for the training.\n", + "\n", + " - `boot_disk_type`: Either SSD or Standard. SSD is faster, and Standard is less expensive. Defaults to SSD.\n", + " - `boot_disk_size_gb`: Size of disk in GB." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "outputs": [], + "source": [ + "DISK_TYPE = \"pd-ssd\" # [ pd-ssd, pd-standard]\n", + "DISK_SIZE = 200 # GB\n", + "\n", + "disk_spec = {\"boot_disk_type\": DISK_TYPE, \"boot_disk_size_gb\": DISK_SIZE}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container" + }, + "source": [ + "### Define the worker pool specification\n", + "\n", + "Next, you define the worker pool specification for your custom training job. The worker pool specification will consist of the following:\n", + "\n", + "- `replica_count`: The number of instances to provision of this machine type.\n", + "- `machine_spec`: The hardware specification.\n", + "- `disk_spec` : (optional) The disk storage specification.\n", + "\n", + "- `python_package`: The Python training package to install on the VM instance(s) and which Python module to invoke, along with command line arguments for the Python module.\n", + "\n", + "Let's dive deeper now into the python package specification:\n", + "\n", + "-`executor_image_spec`: This is the docker image which is configured for your custom training job.\n", + "\n", + "-`package_uris`: This is a list of the locations (URIs) of your python training packages to install on the provisioned instance. The locations need to be in a Cloud Storage bucket. These can be either individual python files or a zip (archive) of an entire package. In the later case, the job service will unzip (unarchive) the contents into the docker image.\n", + "\n", + "-`python_module`: The Python module (script) to invoke for running the custom training job. In this example, you will be invoking `trainer.task.py` -- note that it was not neccessary to append the `.py` suffix.\n", + "\n", + "-`args`: The command line arguments to pass to the corresponding Pythom module. In this example, you will be setting:\n", + " - `\"--model-dir=\" + MODEL_DIR` : The Cloud Storage location where to store the model artifacts. There are two ways to tell the training script where to save the model artifacts:\n", + " - direct: You pass the Cloud Storage location as a command line argument to your training script (set variable `DIRECT = True`), or\n", + " - indirect: The service passes the Cloud Storage location as the environment variable `AIP_MODEL_DIR` to your training script (set variable `DIRECT = False`). In this case, you tell the service the model artifact location in the job specification.\n", + " - `\"--epochs=\" + EPOCHS`: The number of epochs for training.\n", + " - `\"--steps=\" + STEPS`: The number of steps (batches) per epoch.\n", + " - `\"--distribute=\" + TRAIN_STRATEGY\"` : The training distribution strategy to use for single or distributed training.\n", + " - `\"single\"`: single device.\n", + " - `\"mirror\"`: all GPU devices on a single compute instance.\n", + " - `\"multi\"`: all GPU devices on all compute instances." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_job_\" + TIMESTAMP\n", + "MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, JOB_NAME)\n", + "\n", + "if not TRAIN_NGPU or TRAIN_NGPU < 2:\n", + " TRAIN_STRATEGY = \"single\"\n", + "else:\n", + " TRAIN_STRATEGY = \"mirror\"\n", + "\n", + "EPOCHS = 20\n", + "STEPS = 100\n", + "\n", + "DIRECT = True\n", + "if DIRECT:\n", + " CMDARGS = [\n", + " \"--model-dir=\" + MODEL_DIR,\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "else:\n", + " CMDARGS = [\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "\n", + "worker_pool_spec = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": machine_spec,\n", + " \"disk_spec\": disk_spec,\n", + " \"python_package_spec\": {\n", + " \"executor_image_uri\": TRAIN_IMAGE,\n", + " \"package_uris\": [BUCKET_NAME + \"/trainer_imdb.tar.gz\"],\n", + " \"python_module\": \"trainer.task\",\n", + " \"args\": CMDARGS,\n", + " },\n", + " }\n", + "]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "assemble_custom_job_specification" + }, + "source": [ + "### Assemble a job specification\n", + "\n", + "Now assemble the complete description for the custom job specification:\n", + "\n", + "- `display_name`: The human readable name you assign to this custom job.\n", + "- `job_spec`: The specification for the custom job.\n", + " - `worker_pool_specs`: The specification for the machine VM instances.\n", + " - `base_output_directory`: This tells the service the Cloud Storage location where to save the model artifacts (when variable `DIRECT = False`). The service will then pass the location to the training script as the environment variable `AIP_MODEL_DIR`, and the path will be of the form:\n", + "\n", + " /model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "assemble_custom_job_specification" + }, + "outputs": [], + "source": [ + "if DIRECT:\n", + " job_spec = {\"worker_pool_specs\": worker_pool_spec}\n", + "else:\n", + " job_spec = {\n", + " \"worker_pool_specs\": worker_pool_spec,\n", + " \"base_output_directory\": {\"output_uri_prefix\": MODEL_DIR},\n", + " }\n", + "\n", + "custom_job = {\"display_name\": JOB_NAME, \"job_spec\": job_spec}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "examine_training_package" + }, + "source": [ + "### Examine the training package\n", + "\n", + "#### Package layout\n", + "\n", + "Before you start the training, you will look at how a Python package is assembled for a custom training job. When unarchived, the package contains the following directory/file layout.\n", + "\n", + "- PKG-INFO\n", + "- README.md\n", + "- setup.cfg\n", + "- setup.py\n", + "- trainer\n", + " - \\_\\_init\\_\\_.py\n", + " - task.py\n", + "\n", + "The files `setup.cfg` and `setup.py` are the instructions for installing the package into the operating environment of the Docker image.\n", + "\n", + "The file `trainer/task.py` is the Python script for executing the custom training job. *Note*, when we referred to it in the worker pool specification, we replace the directory slash with a dot (`trainer.task`) and dropped the file suffix (`.py`).\n", + "\n", + "#### Package Assembly\n", + "\n", + "In the following cells, you will assemble the training package." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "examine_training_package" + }, + "outputs": [], + "source": [ + "# Make folder for Python training script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\ntag_build =\\n\\ntag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'tensorflow_datasets==1.3.0',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\nName: IMDB Movie Reviews text binary classification\\n\\nVersion: 0.0.0\\n\\nSummary: Demostration training script\\n\\nHome-page: www.google.com\\n\\nAuthor: Google\\n\\nAuthor-email: aferlitsch@google.com\\n\\nLicense: Public\\n\\nDescription: Demo\\n\\nPlatform: Vertex\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:imdb" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the training script task.py. I won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Gets the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n", + "- Loads IMDB Movie Reviews dataset from TF Datasets (tfds).\n", + "- Builds a simple RNN model using TF.Keras model API.\n", + "- Compiles the model (`compile()`).\n", + "- Sets a training distribution strategy according to the argument `args.distribute`.\n", + "- Trains the model (`fit()`) with epochs specified by `args.epochs`.\n", + "- Saves the trained model (`save(args.model_dir)`) to the specified model directory." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:imdb" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Single, Mirror and Multi-Machine Distributed Training for IMDB\n", + "\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "import argparse\n", + "import os\n", + "import sys\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv('AIP_MODEL_DIR'), type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=1e-4, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=20, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=100, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "print(device_lib.list_local_devices())\n", + "\n", + "# Single Machine, single compute device\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "# Single Machine, multiple compute device\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "# Multiple Machine, multiple compute device\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "# Multi-worker configuration\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "# Preparing dataset\n", + "BUFFER_SIZE = 10000\n", + "BATCH_SIZE = 64\n", + "\n", + "\n", + "def make_datasets():\n", + " dataset, info = tfds.load('imdb_reviews/subwords8k', with_info=True,\n", + " as_supervised=True)\n", + " train_dataset, test_dataset = dataset['train'], dataset['test']\n", + " encoder = info.features['text'].encoder\n", + "\n", + " padded_shapes = ([None],())\n", + " return train_dataset.shuffle(BUFFER_SIZE).padded_batch(BATCH_SIZE, padded_shapes), encoder\n", + "\n", + "\n", + "train_dataset, encoder = make_datasets()\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_rnn_model(encoder):\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Embedding(encoder.vocab_size, 64),\n", + " tf.keras.layers.Bidirectional(tf.keras.layers.LSTM(64)),\n", + " tf.keras.layers.Dense(64, activation='relu'),\n", + " tf.keras.layers.Dense(1, activation='sigmoid')\n", + " ])\n", + " model.compile(loss=tf.keras.losses.BinaryCrossentropy(from_logits=True),\n", + " optimizer=tf.keras.optimizers.Adam(args.lr),\n", + " metrics=['accuracy'])\n", + " return model\n", + "\n", + "\n", + "with strategy.scope():\n", + " # Creation of dataset, and model building/compiling need to be within\n", + " # `strategy.scope()`.\n", + " model = build_and_compile_rnn_model(encoder)\n", + "\n", + "# Train the model\n", + "model.fit(train_dataset, epochs=args.epochs, steps_per_epoch=args.steps)\n", + "model.save(args.model_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tarball_training_script" + }, + "source": [ + "#### Store training script on your Cloud Storage bucket\n", + "\n", + "Next, you package the training folder into a compressed tar ball, and then store it in your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tarball_training_script" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_imdb.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job" + }, + "source": [ + "### Train the model\n", + "\n", + "\n", + "Now start the training of your custom training job on Vertex. Use this helper function `create_custom_job`, which takes the following parameter:\n", + "\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "The helper function calls job client service's `create_custom_job` method, with the following parameters:\n", + "\n", + "-`parent`: The Vertex location path to `Dataset`, `Model` and `Endpoint` resources.\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "You will display a handful of the fields returned in `response` object, with the two that are of most interest are:\n", + "\n", + "`response.name`: The Vertex fully qualified identifier assigned to this custom training job. You save this identifier for using in subsequent steps.\n", + "\n", + "`response.state`: The current state of the custom training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job" + }, + "outputs": [], + "source": [ + "def create_custom_job(custom_job):\n", + " response = clients[\"job\"].create_custom_job(parent=PARENT, custom_job=custom_job)\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = create_custom_job(custom_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "job_id:response" + }, + "source": [ + "Now get the unique identifier for the custom job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom job\n", + "job_id = response.name\n", + "# The short numeric ID for the custom job\n", + "job_short_id = job_id.split(\"/\")[-1]\n", + "\n", + "print(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_custom_job" + }, + "source": [ + "### Get information on a custom job\n", + "\n", + "Next, use this helper function `get_custom_job`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "The helper function calls the job client service's`get_custom_job` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "If you recall, you got the Vertex fully qualified identifier for the custom job in the `response.name` field when you called the `create_custom_job` method, and saved the identifier in the variable `job_id`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_custom_job" + }, + "outputs": [], + "source": [ + "def get_custom_job(name, silent=False):\n", + " response = clients[\"job\"].get_custom_job(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = get_custom_job(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete:custom" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, we will need to know the location of the saved model, which the Python script saved in your local Cloud Storage bucket at `MODEL_DIR + '/saved_model.pb'`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete:custom" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_custom_job(job_id, True)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_path_to_deploy = None\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " if not DIRECT:\n", + " MODEL_DIR = MODEL_DIR + \"/model\"\n", + " model_path_to_deploy = MODEL_DIR\n", + " print(\"Training Time:\", response.update_time - response.create_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model_to_deploy:\", model_path_to_deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "load_saved_model" + }, + "source": [ + "## Load the saved model\n", + "\n", + "Your model is stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Now load it from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n", + "\n", + "To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "load_saved_model" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:text" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now let's find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load the IMDB Movie Review test (holdout) data from `tfds.datasets`, using the method `load()`. This will return the dataset as a tuple of two elements. The first element is the dataset and the second is information on the dataset, which will contain the predefined vocabulary encoder. The encoder will convert words into a numerical embedding, which was pretrained and used in the custom training script.\n", + "\n", + "\n", + "When you trained the model, you needed to set a fix input length for your text. For forward feeding batches, the `padded_batch()` property of the corresponding `tf.dataset` was set to pad each input sequence into the same shape for a batch.\n", + "\n", + "For the test data, you also need to set the `padded_batch()` property accordingly." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "evaluate_custom_model:text,imdb" + }, + "outputs": [], + "source": [ + "import tensorflow_datasets as tfds\n", + "\n", + "dataset, info = tfds.load(\"imdb_reviews/subwords8k\", with_info=True, as_supervised=True)\n", + "test_dataset = dataset[\"test\"]\n", + "encoder = info.features[\"text\"].encoder\n", + "\n", + "BATCH_SIZE = 64\n", + "padded_shapes = ([None], ())\n", + "test_dataset = test_dataset.padded_batch(BATCH_SIZE, padded_shapes)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "perform_evaluation_custom:tfds" + }, + "source": [ + "### Perform the model evaluation\n", + "\n", + "Now evaluate how well the model in the custom job did." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "perform_evaluation_custom:tfds" + }, + "outputs": [], + "source": [ + "model.evaluate(test_dataset)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "how_serving_function_works" + }, + "source": [ + "## Upload the model for serving\n", + "\n", + "Next, you will upload your TF.Keras model from the custom job to Vertex `Model` service, which will create a Vertex `Model` resource for your custom model. During upload, you need to define a serving function to convert data to the format your model expects. If you send encoded data to Vertex, your serving function ensures that the data is decoded on the model server before it is passed as input to your model.\n", + "\n", + "### How does the serving function work\n", + "\n", + "When you send a request to an online prediction server, the request is received by a HTTP server. The HTTP server extracts the prediction request from the HTTP request content body. The extracted prediction request is forwarded to the serving function. For Google pre-built prediction containers, the request content is passed to the serving function as a `tf.string`.\n", + "\n", + "The serving function consists of two parts:\n", + "\n", + "- `preprocessing function`:\n", + " - Converts the input (`tf.string`) to the input shape and data type of the underlying model (dynamic graph).\n", + " - Performs the same preprocessing of the data that was done during training the underlying model -- e.g., normalizing, scaling, etc.\n", + "- `post-processing function`:\n", + " - Converts the model output to format expected by the receiving application -- e.q., compresses the output.\n", + " - Packages the output for the the receiving application -- e.g., add headings, make JSON object, etc.\n", + "\n", + "Both the preprocessing and post-processing functions are converted to static graphs which are fused to the model. The output from the underlying model is passed to the post-processing function. The post-processing function passes the converted/packaged output back to the HTTP server. The HTTP server returns the output as the HTTP response content.\n", + "\n", + "One consideration you need to consider when building serving functions for TF.Keras models is that they run as static graphs. That means, you cannot use TF graph operations that require a dynamic graph. If you do, you will get an error during the compile of the serving function which will indicate that you are using an EagerTensor which is not supported." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_signature" + }, + "source": [ + "## Get the serving function signature\n", + "\n", + "You can get the signatures of your model's input and output layers by reloading the model into memory, and querying it for the signatures corresponding to each layer.\n", + "\n", + "When making a prediction request, you need to route the request to the serving function instead of the model, so you need to know the input layer name of the serving function -- which you will use later when you make a prediction request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_signature" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_path_to_deploy)\n", + "\n", + "serving_input = list(\n", + " loaded.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "print(\"Serving function input:\", serving_input)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "upload_the_model" + }, + "source": [ + "### Upload the model\n", + "\n", + "Use this helper function `upload_model` to upload your model, stored in SavedModel format, up to the `Model` service, which will instantiate a Vertex `Model` resource instance for your model. Once you've done that, you can use the `Model` resource instance in the same way as any other Vertex `Model` resource instance, such as deploying to an `Endpoint` resource for serving predictions.\n", + "\n", + "The helper function takes the following parameters:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` service.\n", + "- `image_uri`: The container image for the model deployment.\n", + "- `model_uri`: The Cloud Storage path to our SavedModel artificat. For this tutorial, this is the Cloud Storage location where the `trainer/task.py` saved the model artifacts, which we specified in the variable `MODEL_DIR`.\n", + "\n", + "The helper function calls the `Model` client service's method `upload_model`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for `Dataset`, `Model` and `Endpoint` resources.\n", + "- `model`: The specification for the Vertex `Model` resource instance.\n", + "\n", + "Let's now dive deeper into the Vertex model specification `model`. This is a dictionary object that consists of the following fields:\n", + "\n", + "- `display_name`: A human readable name for the `Model` resource.\n", + "- `metadata_schema_uri`: Since your model was built without an Vertex `Dataset` resource, you will leave this blank (`''`).\n", + "- `artificat_uri`: The Cloud Storage path where the model is stored in SavedModel format.\n", + "- `container_spec`: This is the specification for the Docker container that will be installed on the `Endpoint` resource, from which the `Model` resource will serve predictions. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + "\n", + "Uploading a model into a Vertex Model resource returns a long running operation, since it may take a few moments. You call response.result(), which is a synchronous call and will return when the Vertex Model resource is ready.\n", + "\n", + "The helper function returns the Vertex fully qualified identifier for the corresponding Vertex Model instance upload_model_response.model. You will save the identifier for subsequent steps in the variable model_to_deploy_id." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "upload_the_model" + }, + "outputs": [], + "source": [ + "IMAGE_URI = DEPLOY_IMAGE\n", + "\n", + "\n", + "def upload_model(display_name, image_uri, model_uri):\n", + " model = {\n", + " \"display_name\": display_name,\n", + " \"metadata_schema_uri\": \"\",\n", + " \"artifact_uri\": model_uri,\n", + " \"container_spec\": {\n", + " \"image_uri\": image_uri,\n", + " \"command\": [],\n", + " \"args\": [],\n", + " \"env\": [{\"name\": \"env_name\", \"value\": \"env_value\"}],\n", + " \"ports\": [{\"container_port\": 8080}],\n", + " \"predict_route\": \"\",\n", + " \"health_route\": \"\",\n", + " },\n", + " }\n", + " response = clients[\"model\"].upload_model(parent=PARENT, model=model)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " upload_model_response = response.result(timeout=180)\n", + " print(\"upload_model_response\")\n", + " print(\" model:\", upload_model_response.model)\n", + " return upload_model_response.model\n", + "\n", + "\n", + "model_to_deploy_id = upload_model(\"imdb-\" + TIMESTAMP, IMAGE_URI, model_path_to_deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_model" + }, + "source": [ + "### Get `Model` resource information\n", + "\n", + "Now let's get the model information for just your model. Use this helper function `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource.\n", + "\n", + "This helper function calls the Vertex `Model` client service's method `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_model" + }, + "outputs": [], + "source": [ + "def get_model(name):\n", + " response = clients[\"model\"].get_model(name=name)\n", + " print(response)\n", + "\n", + "\n", + "get_model(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:custom" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex custom `Model` resource. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"imdb_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:dedicated" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `dedicated_resources`: This refers to how many compute instances (replicas) that are scaled for serving prediction requests.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `min_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:dedicated" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"imdb_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " \"machine_spec\": machine_spec,\n", + " },\n", + " \"disable_container_logging\": False,\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "prepare_test_item:text" + }, + "source": [ + "### Prepare the request content\n", + "\n", + "Since the dataset is a `tf.dataset`, which acts as a generator, we must use it as an iterator to access the data items in the test data. We do the following to get a single data item from the test data:\n", + "\n", + "- Set the property for the number of batches to draw per iteration to one using the method `take(1)`.\n", + "- Iterate once through the test data -- i.e., we do a break within the for loop.\n", + "- In the single iteration, we save the data item which is in the form of a tuple.\n", + "- The data item will be the first element of the tuple, which you then will convert from an tensor to a numpy array -- `data[0].numpy()`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "prepare_test_item:text" + }, + "outputs": [], + "source": [ + "import tensorflow_datasets as tfds\n", + "\n", + "dataset, info = tfds.load(\"imdb_reviews/subwords8k\", with_info=True, as_supervised=True)\n", + "test_dataset = dataset[\"test\"]\n", + "\n", + "test_dataset.take(1)\n", + "for data in test_dataset:\n", + " print(data)\n", + " break\n", + "\n", + "test_item = data[0].numpy()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "send_prediction_request:text" + }, + "source": [ + "### Send the prediction request\n", + "\n", + "Ok, now you have a test data item. Use this helper function `predict_data`, which takes the following parameters:\n", + "\n", + "- `data`: The test data item is a 64 padded numpy 1D array.\n", + "- `endpoint`: The Vertex AI fully qualified identifier for the endpoint where the model was deployed.\n", + "- `parameters_dict`: Additional parameters for serving.\n", + "\n", + "This function uses the prediction client service and calls the `predict` method with the following parameters:\n", + "\n", + "- `endpoint`: The Vertex AI fully qualified identifier for the endpoint where the model was deployed.\n", + "- `instances`: A list of instances (data items) to predict.\n", + "- `parameters`: Additional parameters for serving.\n", + "\n", + "To pass the test data to the prediction service, you must package it for transmission to the serving binary as follows:\n", + "\n", + " 1. Convert the data item from a 1D numpy array to a 1D Python list.\n", + " 2. Convert the prediction request to a serialized Google protobuf (`json_format.ParseDict()`)\n", + "\n", + "Each instance in the prediction request is a dictionary entry of the form:\n", + "\n", + " {input_name: content}\n", + "\n", + "- `input_name`: the name of the input layer of the underlying model.\n", + "- `content`: The data item as a 1D Python list.\n", + "\n", + "Since the `predict()` service can take multiple data items (instances), you will send your single data item as a list of one data item. As a final step, you package the instances list into Google's protobuf format -- which is what we pass to the `predict()` service.\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding image in the request. You will see in the output for each prediction:\n", + "\n", + "- `predictions` -- the predicated binary sentiment between 0 (negative) and 1 (positive)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "send_prediction_request:text" + }, + "outputs": [], + "source": [ + "def predict_data(data, endpoint, parameters_dict):\n", + " parameters = json_format.ParseDict(parameters_dict, Value())\n", + "\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [{serving_input: data.tolist()}]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", prediction)\n", + "\n", + "\n", + "predict_data(test_item, endpoint_id, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_custom_text_binary_classification_online.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_custom_text_binary_classification_online_container.ipynb b/notebooks/community/gapic/custom/showcase_custom_text_binary_classification_online_container.ipynb new file mode 100644 index 000000000..2751ef3f4 --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_custom_text_binary_classification_online_container.ipynb @@ -0,0 +1,2192 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Custom training text binary classification model with custom container for online prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom,training,custom_container" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to train using a custom container and deploy a custom text binary classification model for online prediction." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:custom,imdb,tbn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [IMDB Movie Reviews](https://www.tensorflow.org/datasets/catalog/imdb_reviews) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts whether a review is positive or negative in sentiment." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,training,custom_container,online_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create a custom model from a Python script in a custom Docker container using the Vertex client library, and then do a prediction on the deployed model by sending data. You can alternatively create custom models using `gcloud` command-line tool or online using Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex custom job for training a model.\n", + "- Train a TensorFlow model using a custom container.\n", + "- Retrieve and load the model artifacts.\n", + "- View the model evaluation.\n", + "- Upload the model as a Vertex `Model` resource.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:training,cpu,prediction,cpu" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for training and prediction.\n", + "\n", + "Set the variables `TRAIN_GPU/TRAIN_NGPU` and `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU.\n", + "\n", + "*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3 -- which is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:training,cpu,prediction,cpu" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n", + " TRAIN_GPU, TRAIN_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n", + " )\n", + "else:\n", + " TRAIN_GPU, TRAIN_NGPU = (None, None)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (None, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:prediction" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for prediction\n", + "\n", + "- Set the variable `TF` to the TensorFlow version of the container image. For example, `2-1` would be version 2.1, and `1-15` would be version 1.15. The following list shows some of the pre-built images available:\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-3:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-90:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-82:latest`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-23:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-22:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-20:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for prediction](https://cloud.google.com/vertex-ai/docs/predictions/pre-built-containers)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf2-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf2-cpu.{}\".format(TF)\n", + "else:\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "DEPLOY_IMAGE = \"gcr.io/cloud-aiplatform/prediction/{}:latest\".format(DEPLOY_VERSION)\n", + "\n", + "print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:training,prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for training and prediction.\n", + "\n", + "- Set the variables `TRAIN_COMPUTE` and `DEPLOY_COMPUTE` to configure the compute resources for the VMs you will use for for training and prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: The following is not supported for training:*\n", + "\n", + " - `standard`: 2 vCPUs\n", + " - `highcpu`: 2, 4 and 8 vCPUs\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_TRAIN_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Train machine type\", TRAIN_COMPUTE)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own custom model and training for IMDB Movie Reviews." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:custom" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Model Service for `Model` resources.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:custom" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"job\"] = create_job_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_model" + }, + "source": [ + "## Train a model\n", + "\n", + "There are two ways you can train a custom model using a container image:\n", + "\n", + "- **Use a Google Cloud prebuilt container**. If you use a prebuilt container, you will additionally specify a Python package to install into the container image. This Python package contains your code for training a custom model.\n", + "\n", + "- **Use your own custom container image**. If you use your own container, the container needs to contain your code for training a custom model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_docker_container:training" + }, + "source": [ + "### Create a Docker file\n", + "\n", + "In this tutorial, you train a IMDB Movie Reviews model using your own custom container.\n", + "\n", + "To use your own custom container, you build a Docker file. First, you will create a directory for the container components." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "examine_training_package" + }, + "source": [ + "### Examine the training package\n", + "\n", + "#### Package layout\n", + "\n", + "Before you start the training, you will look at how a Python package is assembled for a custom training job. When unarchived, the package contains the following directory/file layout.\n", + "\n", + "- PKG-INFO\n", + "- README.md\n", + "- setup.cfg\n", + "- setup.py\n", + "- trainer\n", + " - \\_\\_init\\_\\_.py\n", + " - task.py\n", + "\n", + "The files `setup.cfg` and `setup.py` are the instructions for installing the package into the operating environment of the Docker image.\n", + "\n", + "The file `trainer/task.py` is the Python script for executing the custom training job. *Note*, when we referred to it in the worker pool specification, we replace the directory slash with a dot (`trainer.task`) and dropped the file suffix (`.py`).\n", + "\n", + "#### Package Assembly\n", + "\n", + "In the following cells, you will assemble the training package." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "examine_training_package" + }, + "outputs": [], + "source": [ + "# Make folder for Python training script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\ntag_build =\\n\\ntag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'tensorflow_datasets==1.3.0',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\nName: IMDB Movie Reviews text binary classification\\n\\nVersion: 0.0.0\\n\\nSummary: Demostration training script\\n\\nHome-page: www.google.com\\n\\nAuthor: Google\\n\\nAuthor-email: aferlitsch@google.com\\n\\nLicense: Public\\n\\nDescription: Demo\\n\\nPlatform: Vertex\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:imdb" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the training script task.py. I won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Gets the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n", + "- Loads IMDB Movie Reviews dataset from TF Datasets (tfds).\n", + "- Builds a simple RNN model using TF.Keras model API.\n", + "- Compiles the model (`compile()`).\n", + "- Sets a training distribution strategy according to the argument `args.distribute`.\n", + "- Trains the model (`fit()`) with epochs specified by `args.epochs`.\n", + "- Saves the trained model (`save(args.model_dir)`) to the specified model directory." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:imdb" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Single, Mirror and Multi-Machine Distributed Training for IMDB\n", + "\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "import argparse\n", + "import os\n", + "import sys\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv('AIP_MODEL_DIR'), type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=1e-4, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=20, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=100, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "print(device_lib.list_local_devices())\n", + "\n", + "# Single Machine, single compute device\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "# Single Machine, multiple compute device\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "# Multiple Machine, multiple compute device\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "# Multi-worker configuration\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "# Preparing dataset\n", + "BUFFER_SIZE = 10000\n", + "BATCH_SIZE = 64\n", + "\n", + "\n", + "def make_datasets():\n", + " dataset, info = tfds.load('imdb_reviews/subwords8k', with_info=True,\n", + " as_supervised=True)\n", + " train_dataset, test_dataset = dataset['train'], dataset['test']\n", + " encoder = info.features['text'].encoder\n", + "\n", + " padded_shapes = ([None],())\n", + " return train_dataset.shuffle(BUFFER_SIZE).padded_batch(BATCH_SIZE, padded_shapes), encoder\n", + "\n", + "\n", + "train_dataset, encoder = make_datasets()\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_rnn_model(encoder):\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Embedding(encoder.vocab_size, 64),\n", + " tf.keras.layers.Bidirectional(tf.keras.layers.LSTM(64)),\n", + " tf.keras.layers.Dense(64, activation='relu'),\n", + " tf.keras.layers.Dense(1, activation='sigmoid')\n", + " ])\n", + " model.compile(loss=tf.keras.losses.BinaryCrossentropy(from_logits=True),\n", + " optimizer=tf.keras.optimizers.Adam(args.lr),\n", + " metrics=['accuracy'])\n", + " return model\n", + "\n", + "\n", + "with strategy.scope():\n", + " # Creation of dataset, and model building/compiling need to be within\n", + " # `strategy.scope()`.\n", + " model = build_and_compile_rnn_model(encoder)\n", + "\n", + "# Train the model\n", + "model.fit(train_dataset, epochs=args.epochs, steps_per_epoch=args.steps)\n", + "model.save(args.model_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "write_docker_file:training,tf-dlvm" + }, + "source": [ + "#### Write the Docker file contents\n", + "\n", + "Your first step in containerizing your code is to create a Docker file. In your Docker you’ll include all the commands needed to run your container image. It’ll install all the libraries you’re using and set up the entry point for your training code.\n", + "\n", + "1. Install a pre-defined container image from TensorFlow repository for deep learning images.\n", + "2. Copies in the Python training code, to be shown subsequently.\n", + "3. Sets the entry into the Python training script as `trainer/task.py`. Note, the `.py` is dropped in the ENTRYPOINT command, as it is implied." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "write_docker_file:training,tf-dlvm" + }, + "outputs": [], + "source": [ + "%%writefile custom/Dockerfile\n", + "\n", + "FROM gcr.io/deeplearning-platform-release/tf2-cpu.2-1\n", + "WORKDIR /root\n", + "\n", + "WORKDIR /\n", + "\n", + "# Copies the trainer code to the docker image.\n", + "COPY trainer /trainer\n", + "\n", + "# Sets up the entry point to invoke the trainer.\n", + "ENTRYPOINT [\"python\", \"-m\", \"trainer.task\"]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "name_container:training" + }, + "source": [ + "#### Build the container locally\n", + "\n", + "Next, you will provide a name for your customer container that you will use when you submit it to the Google Container Registry." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "name_container:training" + }, + "outputs": [], + "source": [ + "TRAIN_IMAGE = \"gcr.io/\" + PROJECT_ID + \"/imdb:v1\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "build_container:training" + }, + "source": [ + "Next, build the container." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "build_container:training" + }, + "outputs": [], + "source": [ + "! docker build custom -t $TRAIN_IMAGE" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "test_container:training" + }, + "source": [ + "#### Test the container locally\n", + "\n", + "Run the container within your notebook instance to ensure it’s working correctly. You will run it for 5 epochs." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "test_container:training" + }, + "outputs": [], + "source": [ + "! docker run $TRAIN_IMAGE --epochs=5" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "register_container:training" + }, + "source": [ + "#### Register the custom container\n", + "\n", + "When you’ve finished running the container locally, push it to Google Container Registry." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "register_container:training" + }, + "outputs": [], + "source": [ + "! docker push $TRAIN_IMAGE" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tarball_training_script" + }, + "source": [ + "#### Store training script on your Cloud Storage bucket\n", + "\n", + "Next, you package the training folder into a compressed tar ball, and then store it in your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tarball_training_script" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_imdb.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_specification:custom_container" + }, + "source": [ + "## Prepare your custom job specification\n", + "\n", + "Now that your clients are ready, your first step is to create a Job Specification for your custom training job. The job specification will consist of the following:\n", + "\n", + "- `worker_pool_spec` : The specification of the type of machine(s) you will use for training and how many (single or distributed)\n", + "- `container_spec` : The specification of the custom container." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "source": [ + "### Prepare your machine specification\n", + "\n", + "Now define the machine specification for your custom training job. This tells Vertex what type of machine instance to provision for the training.\n", + " - `machine_type`: The type of GCP instance to provision -- e.g., n1-standard-8.\n", + " - `accelerator_type`: The type, if any, of hardware accelerator. In this tutorial if you previously set the variable `TRAIN_GPU != None`, you are using a GPU; otherwise you will use a CPU.\n", + " - `accelerator_count`: The number of accelerators." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "outputs": [], + "source": [ + "if TRAIN_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": TRAIN_COMPUTE,\n", + " \"accelerator_type\": TRAIN_GPU,\n", + " \"accelerator_count\": TRAIN_NGPU,\n", + " }\n", + "else:\n", + " machine_spec = {\"machine_type\": TRAIN_COMPUTE, \"accelerator_count\": 0}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "source": [ + "### Prepare your disk specification\n", + "\n", + "(optional) Now define the disk specification for your custom training job. This tells Vertex what type and size of disk to provision in each machine instance for the training.\n", + "\n", + " - `boot_disk_type`: Either SSD or Standard. SSD is faster, and Standard is less expensive. Defaults to SSD.\n", + " - `boot_disk_size_gb`: Size of disk in GB." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "outputs": [], + "source": [ + "DISK_TYPE = \"pd-ssd\" # [ pd-ssd, pd-standard]\n", + "DISK_SIZE = 200 # GB\n", + "\n", + "disk_spec = {\"boot_disk_type\": DISK_TYPE, \"boot_disk_size_gb\": DISK_SIZE}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "prepare_custom_container_specification" + }, + "source": [ + "### Prepare your container specification\n", + "\n", + "Now define the container specification for your custom training container:\n", + "\n", + "- `image_uri`: The custom container image.\n", + "- `args`: The command-line arguments to pass to the executable that is set as the entry point into the container.\n", + " - `--model-dir` : For our demonstrations, we use this command-line argument to specify where to store the model artifacts.\n", + " - direct: You pass the Cloud Storage location as a command line argument to your training script (set variable `DIRECT = True`), or\n", + " - indirect: The service passes the Cloud Storage location as the environment variable `AIP_MODEL_DIR` to your training script (set variable `DIRECT = False`). In this case, you tell the service the model artifact location in the job specification.\n", + " - `\"--epochs=\" + EPOCHS`: The number of epochs for training.\n", + " - `\"--steps=\" + STEPS`: The number of steps per epoch." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "prepare_custom_container_specification" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"_custom_container\" + TIMESTAMP\n", + "MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, JOB_NAME)\n", + "\n", + "EPOCHS = 20\n", + "STEPS = 100\n", + "\n", + "DIRECT = True\n", + "if DIRECT:\n", + " CMDARGS = [\n", + " \"--model-dir=\" + MODEL_DIR,\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " ]\n", + "else:\n", + " CMDARGS = [\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " ]\n", + "\n", + "container_spec = {\n", + " \"image_uri\": TRAIN_IMAGE,\n", + " \"args\": CMDARGS,\n", + "}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_worker_pool_specification:custom_container" + }, + "source": [ + "### Define the worker pool specification\n", + "\n", + "Next, you define the worker pool specification for your custom training job. The worker pool specification will consist of the following:\n", + "\n", + "- `replica_count`: The number of instances to provision of this machine type.\n", + "- `machine_spec`: The hardware specification.\n", + "- `disk_spec` : (optional) The disk storage specification.\n", + "- `container_spec`: The Docker container to install on the VM instance(s)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_worker_pool_specification:custom_container" + }, + "outputs": [], + "source": [ + "worker_pool_spec = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": machine_spec,\n", + " \"container_spec\": container_spec,\n", + " \"disk_spec\": disk_spec,\n", + " }\n", + "]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "assemble_custom_job_specification" + }, + "source": [ + "### Assemble a job specification\n", + "\n", + "Now assemble the complete description for the custom job specification:\n", + "\n", + "- `display_name`: The human readable name you assign to this custom job.\n", + "- `job_spec`: The specification for the custom job.\n", + " - `worker_pool_specs`: The specification for the machine VM instances.\n", + " - `base_output_directory`: This tells the service the Cloud Storage location where to save the model artifacts (when variable `DIRECT = False`). The service will then pass the location to the training script as the environment variable `AIP_MODEL_DIR`, and the path will be of the form:\n", + "\n", + " /model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "assemble_custom_job_specification" + }, + "outputs": [], + "source": [ + "if DIRECT:\n", + " job_spec = {\"worker_pool_specs\": worker_pool_spec}\n", + "else:\n", + " job_spec = {\n", + " \"worker_pool_specs\": worker_pool_spec,\n", + " \"base_output_directory\": {\"output_uri_prefix\": MODEL_DIR},\n", + " }\n", + "\n", + "custom_job = {\"display_name\": JOB_NAME, \"job_spec\": job_spec}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job" + }, + "source": [ + "### Train the model\n", + "\n", + "\n", + "Now start the training of your custom training job on Vertex. Use this helper function `create_custom_job`, which takes the following parameter:\n", + "\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "The helper function calls job client service's `create_custom_job` method, with the following parameters:\n", + "\n", + "-`parent`: The Vertex location path to `Dataset`, `Model` and `Endpoint` resources.\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "You will display a handful of the fields returned in `response` object, with the two that are of most interest are:\n", + "\n", + "`response.name`: The Vertex fully qualified identifier assigned to this custom training job. You save this identifier for using in subsequent steps.\n", + "\n", + "`response.state`: The current state of the custom training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job" + }, + "outputs": [], + "source": [ + "def create_custom_job(custom_job):\n", + " response = clients[\"job\"].create_custom_job(parent=PARENT, custom_job=custom_job)\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = create_custom_job(custom_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "job_id:response" + }, + "source": [ + "Now get the unique identifier for the custom job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom job\n", + "job_id = response.name\n", + "# The short numeric ID for the custom job\n", + "job_short_id = job_id.split(\"/\")[-1]\n", + "\n", + "print(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_custom_job" + }, + "source": [ + "### Get information on a custom job\n", + "\n", + "Next, use this helper function `get_custom_job`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "The helper function calls the job client service's`get_custom_job` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "If you recall, you got the Vertex fully qualified identifier for the custom job in the `response.name` field when you called the `create_custom_job` method, and saved the identifier in the variable `job_id`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_custom_job" + }, + "outputs": [], + "source": [ + "def get_custom_job(name, silent=False):\n", + " response = clients[\"job\"].get_custom_job(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = get_custom_job(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete:custom" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, we will need to know the location of the saved model, which the Python script saved in your local Cloud Storage bucket at `MODEL_DIR + '/saved_model.pb'`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete:custom" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_custom_job(job_id, True)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_path_to_deploy = None\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " if not DIRECT:\n", + " MODEL_DIR = MODEL_DIR + \"/model\"\n", + " model_path_to_deploy = MODEL_DIR\n", + " print(\"Training Time:\", response.update_time - response.create_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model_to_deploy:\", model_path_to_deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "load_saved_model" + }, + "source": [ + "## Load the saved model\n", + "\n", + "Your model is stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Now load it from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n", + "\n", + "To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "load_saved_model" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:text" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now let's find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load the IMDB Movie Review test (holdout) data from `tfds.datasets`, using the method `load()`. This will return the dataset as a tuple of two elements. The first element is the dataset and the second is information on the dataset, which will contain the predefined vocabulary encoder. The encoder will convert words into a numerical embedding, which was pretrained and used in the custom training script.\n", + "\n", + "\n", + "When you trained the model, you needed to set a fix input length for your text. For forward feeding batches, the `padded_batch()` property of the corresponding `tf.dataset` was set to pad each input sequence into the same shape for a batch.\n", + "\n", + "For the test data, you also need to set the `padded_batch()` property accordingly." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "evaluate_custom_model:text,imdb" + }, + "outputs": [], + "source": [ + "import tensorflow_datasets as tfds\n", + "\n", + "dataset, info = tfds.load(\"imdb_reviews/subwords8k\", with_info=True, as_supervised=True)\n", + "test_dataset = dataset[\"test\"]\n", + "encoder = info.features[\"text\"].encoder\n", + "\n", + "BATCH_SIZE = 64\n", + "padded_shapes = ([None], ())\n", + "test_dataset = test_dataset.padded_batch(BATCH_SIZE, padded_shapes)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "perform_evaluation_custom:tfds" + }, + "source": [ + "### Perform the model evaluation\n", + "\n", + "Now evaluate how well the model in the custom job did." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "perform_evaluation_custom:tfds" + }, + "outputs": [], + "source": [ + "model.evaluate(test_dataset)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "how_serving_function_works" + }, + "source": [ + "## Upload the model for serving\n", + "\n", + "Next, you will upload your TF.Keras model from the custom job to Vertex `Model` service, which will create a Vertex `Model` resource for your custom model. During upload, you need to define a serving function to convert data to the format your model expects. If you send encoded data to Vertex, your serving function ensures that the data is decoded on the model server before it is passed as input to your model.\n", + "\n", + "### How does the serving function work\n", + "\n", + "When you send a request to an online prediction server, the request is received by a HTTP server. The HTTP server extracts the prediction request from the HTTP request content body. The extracted prediction request is forwarded to the serving function. For Google pre-built prediction containers, the request content is passed to the serving function as a `tf.string`.\n", + "\n", + "The serving function consists of two parts:\n", + "\n", + "- `preprocessing function`:\n", + " - Converts the input (`tf.string`) to the input shape and data type of the underlying model (dynamic graph).\n", + " - Performs the same preprocessing of the data that was done during training the underlying model -- e.g., normalizing, scaling, etc.\n", + "- `post-processing function`:\n", + " - Converts the model output to format expected by the receiving application -- e.q., compresses the output.\n", + " - Packages the output for the the receiving application -- e.g., add headings, make JSON object, etc.\n", + "\n", + "Both the preprocessing and post-processing functions are converted to static graphs which are fused to the model. The output from the underlying model is passed to the post-processing function. The post-processing function passes the converted/packaged output back to the HTTP server. The HTTP server returns the output as the HTTP response content.\n", + "\n", + "One consideration you need to consider when building serving functions for TF.Keras models is that they run as static graphs. That means, you cannot use TF graph operations that require a dynamic graph. If you do, you will get an error during the compile of the serving function which will indicate that you are using an EagerTensor which is not supported." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_signature" + }, + "source": [ + "## Get the serving function signature\n", + "\n", + "You can get the signatures of your model's input and output layers by reloading the model into memory, and querying it for the signatures corresponding to each layer.\n", + "\n", + "When making a prediction request, you need to route the request to the serving function instead of the model, so you need to know the input layer name of the serving function -- which you will use later when you make a prediction request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_signature" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_path_to_deploy)\n", + "\n", + "serving_input = list(\n", + " loaded.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "print(\"Serving function input:\", serving_input)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "upload_the_model" + }, + "source": [ + "### Upload the model\n", + "\n", + "Use this helper function `upload_model` to upload your model, stored in SavedModel format, up to the `Model` service, which will instantiate a Vertex `Model` resource instance for your model. Once you've done that, you can use the `Model` resource instance in the same way as any other Vertex `Model` resource instance, such as deploying to an `Endpoint` resource for serving predictions.\n", + "\n", + "The helper function takes the following parameters:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` service.\n", + "- `image_uri`: The container image for the model deployment.\n", + "- `model_uri`: The Cloud Storage path to our SavedModel artificat. For this tutorial, this is the Cloud Storage location where the `trainer/task.py` saved the model artifacts, which we specified in the variable `MODEL_DIR`.\n", + "\n", + "The helper function calls the `Model` client service's method `upload_model`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for `Dataset`, `Model` and `Endpoint` resources.\n", + "- `model`: The specification for the Vertex `Model` resource instance.\n", + "\n", + "Let's now dive deeper into the Vertex model specification `model`. This is a dictionary object that consists of the following fields:\n", + "\n", + "- `display_name`: A human readable name for the `Model` resource.\n", + "- `metadata_schema_uri`: Since your model was built without an Vertex `Dataset` resource, you will leave this blank (`''`).\n", + "- `artificat_uri`: The Cloud Storage path where the model is stored in SavedModel format.\n", + "- `container_spec`: This is the specification for the Docker container that will be installed on the `Endpoint` resource, from which the `Model` resource will serve predictions. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + "\n", + "Uploading a model into a Vertex Model resource returns a long running operation, since it may take a few moments. You call response.result(), which is a synchronous call and will return when the Vertex Model resource is ready.\n", + "\n", + "The helper function returns the Vertex fully qualified identifier for the corresponding Vertex Model instance upload_model_response.model. You will save the identifier for subsequent steps in the variable model_to_deploy_id." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "upload_the_model" + }, + "outputs": [], + "source": [ + "IMAGE_URI = DEPLOY_IMAGE\n", + "\n", + "\n", + "def upload_model(display_name, image_uri, model_uri):\n", + " model = {\n", + " \"display_name\": display_name,\n", + " \"metadata_schema_uri\": \"\",\n", + " \"artifact_uri\": model_uri,\n", + " \"container_spec\": {\n", + " \"image_uri\": image_uri,\n", + " \"command\": [],\n", + " \"args\": [],\n", + " \"env\": [{\"name\": \"env_name\", \"value\": \"env_value\"}],\n", + " \"ports\": [{\"container_port\": 8080}],\n", + " \"predict_route\": \"\",\n", + " \"health_route\": \"\",\n", + " },\n", + " }\n", + " response = clients[\"model\"].upload_model(parent=PARENT, model=model)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " upload_model_response = response.result(timeout=180)\n", + " print(\"upload_model_response\")\n", + " print(\" model:\", upload_model_response.model)\n", + " return upload_model_response.model\n", + "\n", + "\n", + "model_to_deploy_id = upload_model(\"imdb-\" + TIMESTAMP, IMAGE_URI, model_path_to_deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_model" + }, + "source": [ + "### Get `Model` resource information\n", + "\n", + "Now let's get the model information for just your model. Use this helper function `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource.\n", + "\n", + "This helper function calls the Vertex `Model` client service's method `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_model" + }, + "outputs": [], + "source": [ + "def get_model(name):\n", + " response = clients[\"model\"].get_model(name=name)\n", + " print(response)\n", + "\n", + "\n", + "get_model(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:custom" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex custom `Model` resource. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"imdb_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:dedicated" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `dedicated_resources`: This refers to how many compute instances (replicas) that are scaled for serving prediction requests.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `min_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:dedicated" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"imdb_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " \"machine_spec\": machine_spec,\n", + " },\n", + " \"disable_container_logging\": False,\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "prepare_test_item:text" + }, + "source": [ + "### Prepare the request content\n", + "\n", + "Since the dataset is a `tf.dataset`, which acts as a generator, we must use it as an iterator to access the data items in the test data. We do the following to get a single data item from the test data:\n", + "\n", + "- Set the property for the number of batches to draw per iteration to one using the method `take(1)`.\n", + "- Iterate once through the test data -- i.e., we do a break within the for loop.\n", + "- In the single iteration, we save the data item which is in the form of a tuple.\n", + "- The data item will be the first element of the tuple, which you then will convert from an tensor to a numpy array -- `data[0].numpy()`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "prepare_test_item:text" + }, + "outputs": [], + "source": [ + "import tensorflow_datasets as tfds\n", + "\n", + "dataset, info = tfds.load(\"imdb_reviews/subwords8k\", with_info=True, as_supervised=True)\n", + "test_dataset = dataset[\"test\"]\n", + "\n", + "test_dataset.take(1)\n", + "for data in test_dataset:\n", + " print(data)\n", + " break\n", + "\n", + "test_item = data[0].numpy()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "send_prediction_request:text" + }, + "source": [ + "### Send the prediction request\n", + "\n", + "Ok, now you have a test data item. Use this helper function `predict_data`, which takes the following parameters:\n", + "\n", + "- `data`: The test data item is a 64 padded numpy 1D array.\n", + "- `endpoint`: The Vertex AI fully qualified identifier for the endpoint where the model was deployed.\n", + "- `parameters_dict`: Additional parameters for serving.\n", + "\n", + "This function uses the prediction client service and calls the `predict` method with the following parameters:\n", + "\n", + "- `endpoint`: The Vertex AI fully qualified identifier for the endpoint where the model was deployed.\n", + "- `instances`: A list of instances (data items) to predict.\n", + "- `parameters`: Additional parameters for serving.\n", + "\n", + "To pass the test data to the prediction service, you must package it for transmission to the serving binary as follows:\n", + "\n", + " 1. Convert the data item from a 1D numpy array to a 1D Python list.\n", + " 2. Convert the prediction request to a serialized Google protobuf (`json_format.ParseDict()`)\n", + "\n", + "Each instance in the prediction request is a dictionary entry of the form:\n", + "\n", + " {input_name: content}\n", + "\n", + "- `input_name`: the name of the input layer of the underlying model.\n", + "- `content`: The data item as a 1D Python list.\n", + "\n", + "Since the `predict()` service can take multiple data items (instances), you will send your single data item as a list of one data item. As a final step, you package the instances list into Google's protobuf format -- which is what we pass to the `predict()` service.\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding image in the request. You will see in the output for each prediction:\n", + "\n", + "- `predictions` -- the predicated binary sentiment between 0 (negative) and 1 (positive)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "send_prediction_request:text" + }, + "outputs": [], + "source": [ + "def predict_data(data, endpoint, parameters_dict):\n", + " parameters = json_format.ParseDict(parameters_dict, Value())\n", + "\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [{serving_input: data.tolist()}]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", prediction)\n", + "\n", + "\n", + "predict_data(test_item, endpoint_id, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_custom_text_binary_classification_online_container.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_custom_text_binary_classification_online_pipeline.ipynb b/notebooks/community/gapic/custom/showcase_custom_text_binary_classification_online_pipeline.ipynb new file mode 100644 index 000000000..367f8fd89 --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_custom_text_binary_classification_online_pipeline.ipynb @@ -0,0 +1,2159 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Custom training text binary classification model with pipeline for online prediction with training pipeline\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom,training_pipeline" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to train and deploy a custom text binary classification model for online prediction, using a training pipeline." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:custom,imdb,tbn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [IMDB Movie Reviews](https://www.tensorflow.org/datasets/catalog/imdb_reviews) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts whether a review is positive or negative in sentiment." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,training,online_prediction,training_pipeline" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create a custom model from a Python script in a Google prebuilt Docker container using the Vertex client library, and then do a prediction on the deployed model by sending data. You can alternatively create custom models using `gcloud` command-line tool or online using Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex custom job for training a model.\n", + "- Create a `TrainingPipeline` resource.\n", + "- Train a TensorFlow model with the `TrainingPipeline` resource.\n", + "- Retrieve and load the model artifacts.\n", + "- View the model evaluation.\n", + "- Upload the model as a Vertex `Model` resource.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "custom_constants" + }, + "source": [ + "#### CustomJob constants\n", + "\n", + "Set constants unique to CustomJob training:\n", + "\n", + "- Dataset Training Schemas: Tells the `Pipeline` resource service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "custom_constants" + }, + "outputs": [], + "source": [ + "CUSTOM_TASK_GCS_PATH = (\n", + " \"gs://google-cloud-aiplatform/schema/trainingjob/definition/custom_task_1.0.0.yaml\"\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:training,cpu,prediction,cpu" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for training and prediction.\n", + "\n", + "Set the variables `TRAIN_GPU/TRAIN_NGPU` and `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU.\n", + "\n", + "*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3 -- which is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:training,cpu,prediction,cpu" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n", + " TRAIN_GPU, TRAIN_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n", + " )\n", + "else:\n", + " TRAIN_GPU, TRAIN_NGPU = (None, None)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (None, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:training,prediction" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for training and prediction\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-3:latest`\n", + " - TensorFlow 2.4\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-4:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/training/xgboost-cpu.1-1`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/training/scikit-learn-cpu.0-23:latest`\n", + " - Pytorch\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-5:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-6:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-7:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for training](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers).\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-3:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-90:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-82:latest`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-23:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-22:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-20:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for prediction](https://cloud.google.com/vertex-ai/docs/predictions/pre-built-containers)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf2-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf2-cpu.{}\".format(TF)\n", + "else:\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "TRAIN_IMAGE = \"gcr.io/cloud-aiplatform/training/{}:latest\".format(TRAIN_VERSION)\n", + "DEPLOY_IMAGE = \"gcr.io/cloud-aiplatform/prediction/{}:latest\".format(DEPLOY_VERSION)\n", + "\n", + "print(\"Training:\", TRAIN_IMAGE, TRAIN_GPU, TRAIN_NGPU)\n", + "print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU, DEPLOY_NGPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:training,prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for training and prediction.\n", + "\n", + "- Set the variables `TRAIN_COMPUTE` and `DEPLOY_COMPUTE` to configure the compute resources for the VMs you will use for for training and prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: The following is not supported for training:*\n", + "\n", + " - `standard`: 2 vCPUs\n", + " - `highcpu`: 2, 4 and 8 vCPUs\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_TRAIN_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Train machine type\", TRAIN_COMPUTE)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own custom model and training for IMDB Movie Reviews." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:custom,training_pipeline" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Model Service for `Model` resources.\n", + "- Pipeline Service for training.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:custom,training_pipeline" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_model" + }, + "source": [ + "## Train a model\n", + "\n", + "There are two ways you can train a custom model using a container image:\n", + "\n", + "- **Use a Google Cloud prebuilt container**. If you use a prebuilt container, you will additionally specify a Python package to install into the container image. This Python package contains your code for training a custom model.\n", + "\n", + "- **Use your own custom container image**. If you use your own container, the container needs to contain your code for training a custom model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_specification:prebuilt_container" + }, + "source": [ + "## Prepare your custom job specification\n", + "\n", + "Now that your clients are ready, your first step is to create a Job Specification for your custom training job. The job specification will consist of the following:\n", + "\n", + "- `worker_pool_spec` : The specification of the type of machine(s) you will use for training and how many (single or distributed)\n", + "- `python_package_spec` : The specification of the Python package to be installed with the pre-built container." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "source": [ + "### Prepare your machine specification\n", + "\n", + "Now define the machine specification for your custom training job. This tells Vertex what type of machine instance to provision for the training.\n", + " - `machine_type`: The type of GCP instance to provision -- e.g., n1-standard-8.\n", + " - `accelerator_type`: The type, if any, of hardware accelerator. In this tutorial if you previously set the variable `TRAIN_GPU != None`, you are using a GPU; otherwise you will use a CPU.\n", + " - `accelerator_count`: The number of accelerators." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "outputs": [], + "source": [ + "if TRAIN_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": TRAIN_COMPUTE,\n", + " \"accelerator_type\": TRAIN_GPU,\n", + " \"accelerator_count\": TRAIN_NGPU,\n", + " }\n", + "else:\n", + " machine_spec = {\"machine_type\": TRAIN_COMPUTE, \"accelerator_count\": 0}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "source": [ + "### Prepare your disk specification\n", + "\n", + "(optional) Now define the disk specification for your custom training job. This tells Vertex what type and size of disk to provision in each machine instance for the training.\n", + "\n", + " - `boot_disk_type`: Either SSD or Standard. SSD is faster, and Standard is less expensive. Defaults to SSD.\n", + " - `boot_disk_size_gb`: Size of disk in GB." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "outputs": [], + "source": [ + "DISK_TYPE = \"pd-ssd\" # [ pd-ssd, pd-standard]\n", + "DISK_SIZE = 200 # GB\n", + "\n", + "disk_spec = {\"boot_disk_type\": DISK_TYPE, \"boot_disk_size_gb\": DISK_SIZE}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container" + }, + "source": [ + "### Define the worker pool specification\n", + "\n", + "Next, you define the worker pool specification for your custom training job. The worker pool specification will consist of the following:\n", + "\n", + "- `replica_count`: The number of instances to provision of this machine type.\n", + "- `machine_spec`: The hardware specification.\n", + "- `disk_spec` : (optional) The disk storage specification.\n", + "\n", + "- `python_package`: The Python training package to install on the VM instance(s) and which Python module to invoke, along with command line arguments for the Python module.\n", + "\n", + "Let's dive deeper now into the python package specification:\n", + "\n", + "-`executor_image_spec`: This is the docker image which is configured for your custom training job.\n", + "\n", + "-`package_uris`: This is a list of the locations (URIs) of your python training packages to install on the provisioned instance. The locations need to be in a Cloud Storage bucket. These can be either individual python files or a zip (archive) of an entire package. In the later case, the job service will unzip (unarchive) the contents into the docker image.\n", + "\n", + "-`python_module`: The Python module (script) to invoke for running the custom training job. In this example, you will be invoking `trainer.task.py` -- note that it was not neccessary to append the `.py` suffix.\n", + "\n", + "-`args`: The command line arguments to pass to the corresponding Pythom module. In this example, you will be setting:\n", + " - `\"--model-dir=\" + MODEL_DIR` : The Cloud Storage location where to store the model artifacts. There are two ways to tell the training script where to save the model artifacts:\n", + " - direct: You pass the Cloud Storage location as a command line argument to your training script (set variable `DIRECT = True`), or\n", + " - indirect: The service passes the Cloud Storage location as the environment variable `AIP_MODEL_DIR` to your training script (set variable `DIRECT = False`). In this case, you tell the service the model artifact location in the job specification.\n", + " - `\"--epochs=\" + EPOCHS`: The number of epochs for training.\n", + " - `\"--steps=\" + STEPS`: The number of steps (batches) per epoch.\n", + " - `\"--distribute=\" + TRAIN_STRATEGY\"` : The training distribution strategy to use for single or distributed training.\n", + " - `\"single\"`: single device.\n", + " - `\"mirror\"`: all GPU devices on a single compute instance.\n", + " - `\"multi\"`: all GPU devices on all compute instances." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_job_\" + TIMESTAMP\n", + "MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, JOB_NAME)\n", + "\n", + "if not TRAIN_NGPU or TRAIN_NGPU < 2:\n", + " TRAIN_STRATEGY = \"single\"\n", + "else:\n", + " TRAIN_STRATEGY = \"mirror\"\n", + "\n", + "EPOCHS = 20\n", + "STEPS = 100\n", + "\n", + "DIRECT = True\n", + "if DIRECT:\n", + " CMDARGS = [\n", + " \"--model-dir=\" + MODEL_DIR,\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "else:\n", + " CMDARGS = [\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "\n", + "worker_pool_spec = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": machine_spec,\n", + " \"disk_spec\": disk_spec,\n", + " \"python_package_spec\": {\n", + " \"executor_image_uri\": TRAIN_IMAGE,\n", + " \"package_uris\": [BUCKET_NAME + \"/trainer_imdb.tar.gz\"],\n", + " \"python_module\": \"trainer.task\",\n", + " \"args\": CMDARGS,\n", + " },\n", + " }\n", + "]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "examine_training_package" + }, + "source": [ + "### Examine the training package\n", + "\n", + "#### Package layout\n", + "\n", + "Before you start the training, you will look at how a Python package is assembled for a custom training job. When unarchived, the package contains the following directory/file layout.\n", + "\n", + "- PKG-INFO\n", + "- README.md\n", + "- setup.cfg\n", + "- setup.py\n", + "- trainer\n", + " - \\_\\_init\\_\\_.py\n", + " - task.py\n", + "\n", + "The files `setup.cfg` and `setup.py` are the instructions for installing the package into the operating environment of the Docker image.\n", + "\n", + "The file `trainer/task.py` is the Python script for executing the custom training job. *Note*, when we referred to it in the worker pool specification, we replace the directory slash with a dot (`trainer.task`) and dropped the file suffix (`.py`).\n", + "\n", + "#### Package Assembly\n", + "\n", + "In the following cells, you will assemble the training package." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "examine_training_package" + }, + "outputs": [], + "source": [ + "# Make folder for Python training script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\ntag_build =\\n\\ntag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'tensorflow_datasets==1.3.0',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\nName: IMDB Movie Reviews text binary classification\\n\\nVersion: 0.0.0\\n\\nSummary: Demostration training script\\n\\nHome-page: www.google.com\\n\\nAuthor: Google\\n\\nAuthor-email: aferlitsch@google.com\\n\\nLicense: Public\\n\\nDescription: Demo\\n\\nPlatform: Vertex\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:imdb" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the training script task.py. I won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Gets the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n", + "- Loads IMDB Movie Reviews dataset from TF Datasets (tfds).\n", + "- Builds a simple RNN model using TF.Keras model API.\n", + "- Compiles the model (`compile()`).\n", + "- Sets a training distribution strategy according to the argument `args.distribute`.\n", + "- Trains the model (`fit()`) with epochs specified by `args.epochs`.\n", + "- Saves the trained model (`save(args.model_dir)`) to the specified model directory." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:imdb" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Single, Mirror and Multi-Machine Distributed Training for IMDB\n", + "\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "import argparse\n", + "import os\n", + "import sys\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv('AIP_MODEL_DIR'), type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=1e-4, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=20, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=100, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "print(device_lib.list_local_devices())\n", + "\n", + "# Single Machine, single compute device\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "# Single Machine, multiple compute device\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "# Multiple Machine, multiple compute device\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "# Multi-worker configuration\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "# Preparing dataset\n", + "BUFFER_SIZE = 10000\n", + "BATCH_SIZE = 64\n", + "\n", + "\n", + "def make_datasets():\n", + " dataset, info = tfds.load('imdb_reviews/subwords8k', with_info=True,\n", + " as_supervised=True)\n", + " train_dataset, test_dataset = dataset['train'], dataset['test']\n", + " encoder = info.features['text'].encoder\n", + "\n", + " padded_shapes = ([None],())\n", + " return train_dataset.shuffle(BUFFER_SIZE).padded_batch(BATCH_SIZE, padded_shapes), encoder\n", + "\n", + "\n", + "train_dataset, encoder = make_datasets()\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_rnn_model(encoder):\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Embedding(encoder.vocab_size, 64),\n", + " tf.keras.layers.Bidirectional(tf.keras.layers.LSTM(64)),\n", + " tf.keras.layers.Dense(64, activation='relu'),\n", + " tf.keras.layers.Dense(1, activation='sigmoid')\n", + " ])\n", + " model.compile(loss=tf.keras.losses.BinaryCrossentropy(from_logits=True),\n", + " optimizer=tf.keras.optimizers.Adam(args.lr),\n", + " metrics=['accuracy'])\n", + " return model\n", + "\n", + "\n", + "with strategy.scope():\n", + " # Creation of dataset, and model building/compiling need to be within\n", + " # `strategy.scope()`.\n", + " model = build_and_compile_rnn_model(encoder)\n", + "\n", + "# Train the model\n", + "model.fit(train_dataset, epochs=args.epochs, steps_per_epoch=args.steps)\n", + "model.save(args.model_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tarball_training_script" + }, + "source": [ + "#### Store training script on your Cloud Storage bucket\n", + "\n", + "Next, you package the training folder into a compressed tar ball, and then store it in your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tarball_training_script" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_imdb.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_with_pipeline" + }, + "source": [ + "## Train the model using a `TrainingPipeline` resource\n", + "\n", + "Now start training of your custom training job using a training pipeline on Vertex. To train the your custom model, do the following steps:\n", + "\n", + "1. Create a Vertex `TrainingPipeline` resource for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training.\n", + "\n", + "### Create a `TrainingPipeline` resource\n", + "\n", + "You may ask, what do we use a pipeline for? We typically use pipelines when the job (such as training) has multiple steps, generally in sequential order: do step A, do step B, etc. By putting the steps into a pipeline, we gain the benefits of:\n", + "\n", + "1. Being reusable for subsequent training jobs.\n", + "2. Can be containerized and ran as a batch job.\n", + "3. Can be distributed.\n", + "4. All the steps are associated with the same pipeline job for tracking progress.\n", + "\n", + "#### The `training_pipeline` specification\n", + "\n", + "First, you need to describe a pipeline specification. Let's look into the *minimal* requirements for constructing a `training_pipeline` specification for a custom job:\n", + "\n", + "- `display_name`: A human readable name for the pipeline job.\n", + "- `training_task_definition`: The training task schema.\n", + "- `training_task_inputs`: A dictionary describing the requirements for the training job.\n", + "- `model_to_upload`: A dictionary describing the specification for the (uploaded) Vertex custom `Model` resource.\n", + " - `display_name`: A human readable name for the `Model` resource.\n", + " - `artificat_uri`: The Cloud Storage path where the model artifacts are stored in SavedModel format.\n", + " - `container_spec`: This is the specification for the Docker container that will be installed on the `Endpoint` resource, from which the custom model will serve predictions." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_with_pipeline" + }, + "outputs": [], + "source": [ + "from google.protobuf import json_format\n", + "from google.protobuf.struct_pb2 import Value\n", + "\n", + "MODEL_NAME = \"custom_pipeline-\" + TIMESTAMP\n", + "PIPELINE_DISPLAY_NAME = \"custom-training-pipeline\" + TIMESTAMP\n", + "\n", + "training_task_inputs = json_format.ParseDict(\n", + " {\"workerPoolSpecs\": worker_pool_spec}, Value()\n", + ")\n", + "pipeline = {\n", + " \"display_name\": PIPELINE_DISPLAY_NAME,\n", + " \"training_task_definition\": CUSTOM_TASK_GCS_PATH,\n", + " \"training_task_inputs\": training_task_inputs,\n", + " \"model_to_upload\": {\n", + " \"display_name\": PIPELINE_DISPLAY_NAME + \"-model\",\n", + " \"artifact_uri\": MODEL_DIR,\n", + " \"container_spec\": {\"image_uri\": DEPLOY_IMAGE},\n", + " },\n", + "}\n", + "\n", + "print(pipeline)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_training_pipeline:custom" + }, + "source": [ + "#### Create the training pipeline\n", + "\n", + "Use this helper function `create_pipeline`, which takes the following parameter:\n", + "\n", + "- `training_pipeline`: the full specification for the pipeline training job.\n", + "\n", + "The helper function calls the pipeline client service's `create_pipeline` method, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for your `Dataset`, `Model` and `Endpoint` resources.\n", + "- `training_pipeline`: The full specification for the pipeline training job.\n", + "\n", + "The helper function will return the Vertex fully qualified identifier assigned to the training pipeline, which is saved as `pipeline.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_training_pipeline:custom" + }, + "outputs": [], + "source": [ + "def create_pipeline(training_pipeline):\n", + "\n", + " try:\n", + " pipeline = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " )\n", + " print(pipeline)\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + " return pipeline\n", + "\n", + "\n", + "response = create_pipeline(pipeline)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pipeline_id:response" + }, + "source": [ + "Now save the unique identifier of the training pipeline you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pipeline_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the pipeline\n", + "pipeline_id = response.name\n", + "# The short numeric ID for the pipeline\n", + "pipeline_short_id = pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_training_pipeline" + }, + "source": [ + "### Get information on a training pipeline\n", + "\n", + "Now get pipeline information for just this training pipeline instance. The helper function gets the job information for just this job by calling the the job client service's `get_training_pipeline` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified pipeline identifier.\n", + "\n", + "When the model is done training, the pipeline state will be `PIPELINE_STATE_SUCCEEDED`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_training_pipeline" + }, + "outputs": [], + "source": [ + "def get_training_pipeline(name, silent=False):\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"pipeline\")\n", + " print(\" name:\", response.name)\n", + " print(\" display_name:\", response.display_name)\n", + " print(\" state:\", response.state)\n", + " print(\" training_task_definition:\", response.training_task_definition)\n", + " print(\" training_task_inputs:\", dict(response.training_task_inputs))\n", + " print(\" create_time:\", response.create_time)\n", + " print(\" start_time:\", response.start_time)\n", + " print(\" end_time:\", response.end_time)\n", + " print(\" update_time:\", response.update_time)\n", + " print(\" labels:\", dict(response.labels))\n", + " return response\n", + "\n", + "\n", + "response = get_training_pipeline(pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, you will need to know the fully qualified Vertex Model resource identifier, which the pipeline service assigned to it. You can get this from the returned pipeline instance as the field `model_to_deploy.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_training_pipeline(pipeline_id, True)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_id = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " raise Exception(\"Training Job Failed\")\n", + " else:\n", + " model_to_deploy = response.model_to_upload\n", + " model_to_deploy_id = model_to_deploy.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model to deploy:\", model_to_deploy_id)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "model_path_to_deploy" + }, + "outputs": [], + "source": [ + "if not DIRECT:\n", + " MODEL_DIR = MODEL_DIR + \"/model\"\n", + "model_path_to_deploy = MODEL_DIR" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "load_saved_model" + }, + "source": [ + "## Load the saved model\n", + "\n", + "Your model is stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Now load it from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n", + "\n", + "To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "load_saved_model" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:text" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now let's find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load the IMDB Movie Review test (holdout) data from `tfds.datasets`, using the method `load()`. This will return the dataset as a tuple of two elements. The first element is the dataset and the second is information on the dataset, which will contain the predefined vocabulary encoder. The encoder will convert words into a numerical embedding, which was pretrained and used in the custom training script.\n", + "\n", + "\n", + "When you trained the model, you needed to set a fix input length for your text. For forward feeding batches, the `padded_batch()` property of the corresponding `tf.dataset` was set to pad each input sequence into the same shape for a batch.\n", + "\n", + "For the test data, you also need to set the `padded_batch()` property accordingly." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "evaluate_custom_model:text,imdb" + }, + "outputs": [], + "source": [ + "import tensorflow_datasets as tfds\n", + "\n", + "dataset, info = tfds.load(\"imdb_reviews/subwords8k\", with_info=True, as_supervised=True)\n", + "test_dataset = dataset[\"test\"]\n", + "encoder = info.features[\"text\"].encoder\n", + "\n", + "BATCH_SIZE = 64\n", + "padded_shapes = ([None], ())\n", + "test_dataset = test_dataset.padded_batch(BATCH_SIZE, padded_shapes)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "perform_evaluation_custom:tfds" + }, + "source": [ + "### Perform the model evaluation\n", + "\n", + "Now evaluate how well the model in the custom job did." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "perform_evaluation_custom:tfds" + }, + "outputs": [], + "source": [ + "model.evaluate(test_dataset)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "how_serving_function_works" + }, + "source": [ + "## Upload the model for serving\n", + "\n", + "Next, you will upload your TF.Keras model from the custom job to Vertex `Model` service, which will create a Vertex `Model` resource for your custom model. During upload, you need to define a serving function to convert data to the format your model expects. If you send encoded data to Vertex, your serving function ensures that the data is decoded on the model server before it is passed as input to your model.\n", + "\n", + "### How does the serving function work\n", + "\n", + "When you send a request to an online prediction server, the request is received by a HTTP server. The HTTP server extracts the prediction request from the HTTP request content body. The extracted prediction request is forwarded to the serving function. For Google pre-built prediction containers, the request content is passed to the serving function as a `tf.string`.\n", + "\n", + "The serving function consists of two parts:\n", + "\n", + "- `preprocessing function`:\n", + " - Converts the input (`tf.string`) to the input shape and data type of the underlying model (dynamic graph).\n", + " - Performs the same preprocessing of the data that was done during training the underlying model -- e.g., normalizing, scaling, etc.\n", + "- `post-processing function`:\n", + " - Converts the model output to format expected by the receiving application -- e.q., compresses the output.\n", + " - Packages the output for the the receiving application -- e.g., add headings, make JSON object, etc.\n", + "\n", + "Both the preprocessing and post-processing functions are converted to static graphs which are fused to the model. The output from the underlying model is passed to the post-processing function. The post-processing function passes the converted/packaged output back to the HTTP server. The HTTP server returns the output as the HTTP response content.\n", + "\n", + "One consideration you need to consider when building serving functions for TF.Keras models is that they run as static graphs. That means, you cannot use TF graph operations that require a dynamic graph. If you do, you will get an error during the compile of the serving function which will indicate that you are using an EagerTensor which is not supported." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_signature" + }, + "source": [ + "## Get the serving function signature\n", + "\n", + "You can get the signatures of your model's input and output layers by reloading the model into memory, and querying it for the signatures corresponding to each layer.\n", + "\n", + "When making a prediction request, you need to route the request to the serving function instead of the model, so you need to know the input layer name of the serving function -- which you will use later when you make a prediction request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_signature" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_path_to_deploy)\n", + "\n", + "serving_input = list(\n", + " loaded.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "print(\"Serving function input:\", serving_input)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "upload_the_model" + }, + "source": [ + "### Upload the model\n", + "\n", + "Use this helper function `upload_model` to upload your model, stored in SavedModel format, up to the `Model` service, which will instantiate a Vertex `Model` resource instance for your model. Once you've done that, you can use the `Model` resource instance in the same way as any other Vertex `Model` resource instance, such as deploying to an `Endpoint` resource for serving predictions.\n", + "\n", + "The helper function takes the following parameters:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` service.\n", + "- `image_uri`: The container image for the model deployment.\n", + "- `model_uri`: The Cloud Storage path to our SavedModel artificat. For this tutorial, this is the Cloud Storage location where the `trainer/task.py` saved the model artifacts, which we specified in the variable `MODEL_DIR`.\n", + "\n", + "The helper function calls the `Model` client service's method `upload_model`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for `Dataset`, `Model` and `Endpoint` resources.\n", + "- `model`: The specification for the Vertex `Model` resource instance.\n", + "\n", + "Let's now dive deeper into the Vertex model specification `model`. This is a dictionary object that consists of the following fields:\n", + "\n", + "- `display_name`: A human readable name for the `Model` resource.\n", + "- `metadata_schema_uri`: Since your model was built without an Vertex `Dataset` resource, you will leave this blank (`''`).\n", + "- `artificat_uri`: The Cloud Storage path where the model is stored in SavedModel format.\n", + "- `container_spec`: This is the specification for the Docker container that will be installed on the `Endpoint` resource, from which the `Model` resource will serve predictions. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + "\n", + "Uploading a model into a Vertex Model resource returns a long running operation, since it may take a few moments. You call response.result(), which is a synchronous call and will return when the Vertex Model resource is ready.\n", + "\n", + "The helper function returns the Vertex fully qualified identifier for the corresponding Vertex Model instance upload_model_response.model. You will save the identifier for subsequent steps in the variable model_to_deploy_id." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "upload_the_model" + }, + "outputs": [], + "source": [ + "IMAGE_URI = DEPLOY_IMAGE\n", + "\n", + "\n", + "def upload_model(display_name, image_uri, model_uri):\n", + " model = {\n", + " \"display_name\": display_name,\n", + " \"metadata_schema_uri\": \"\",\n", + " \"artifact_uri\": model_uri,\n", + " \"container_spec\": {\n", + " \"image_uri\": image_uri,\n", + " \"command\": [],\n", + " \"args\": [],\n", + " \"env\": [{\"name\": \"env_name\", \"value\": \"env_value\"}],\n", + " \"ports\": [{\"container_port\": 8080}],\n", + " \"predict_route\": \"\",\n", + " \"health_route\": \"\",\n", + " },\n", + " }\n", + " response = clients[\"model\"].upload_model(parent=PARENT, model=model)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " upload_model_response = response.result(timeout=180)\n", + " print(\"upload_model_response\")\n", + " print(\" model:\", upload_model_response.model)\n", + " return upload_model_response.model\n", + "\n", + "\n", + "model_to_deploy_id = upload_model(\"imdb-\" + TIMESTAMP, IMAGE_URI, model_path_to_deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_model" + }, + "source": [ + "### Get `Model` resource information\n", + "\n", + "Now let's get the model information for just your model. Use this helper function `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource.\n", + "\n", + "This helper function calls the Vertex `Model` client service's method `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_model" + }, + "outputs": [], + "source": [ + "def get_model(name):\n", + " response = clients[\"model\"].get_model(name=name)\n", + " print(response)\n", + "\n", + "\n", + "get_model(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:custom" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex custom `Model` resource. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"imdb_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:dedicated" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `dedicated_resources`: This refers to how many compute instances (replicas) that are scaled for serving prediction requests.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `min_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:dedicated" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"imdb_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " \"machine_spec\": machine_spec,\n", + " },\n", + " \"disable_container_logging\": False,\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "prepare_test_item:text" + }, + "source": [ + "### Prepare the request content\n", + "\n", + "Since the dataset is a `tf.dataset`, which acts as a generator, we must use it as an iterator to access the data items in the test data. We do the following to get a single data item from the test data:\n", + "\n", + "- Set the property for the number of batches to draw per iteration to one using the method `take(1)`.\n", + "- Iterate once through the test data -- i.e., we do a break within the for loop.\n", + "- In the single iteration, we save the data item which is in the form of a tuple.\n", + "- The data item will be the first element of the tuple, which you then will convert from an tensor to a numpy array -- `data[0].numpy()`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "prepare_test_item:text" + }, + "outputs": [], + "source": [ + "import tensorflow_datasets as tfds\n", + "\n", + "dataset, info = tfds.load(\"imdb_reviews/subwords8k\", with_info=True, as_supervised=True)\n", + "test_dataset = dataset[\"test\"]\n", + "\n", + "test_dataset.take(1)\n", + "for data in test_dataset:\n", + " print(data)\n", + " break\n", + "\n", + "test_item = data[0].numpy()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "send_prediction_request:text" + }, + "source": [ + "### Send the prediction request\n", + "\n", + "Ok, now you have a test data item. Use this helper function `predict_data`, which takes the following parameters:\n", + "\n", + "- `data`: The test data item is a 64 padded numpy 1D array.\n", + "- `endpoint`: The Vertex AI fully qualified identifier for the endpoint where the model was deployed.\n", + "- `parameters_dict`: Additional parameters for serving.\n", + "\n", + "This function uses the prediction client service and calls the `predict` method with the following parameters:\n", + "\n", + "- `endpoint`: The Vertex AI fully qualified identifier for the endpoint where the model was deployed.\n", + "- `instances`: A list of instances (data items) to predict.\n", + "- `parameters`: Additional parameters for serving.\n", + "\n", + "To pass the test data to the prediction service, you must package it for transmission to the serving binary as follows:\n", + "\n", + " 1. Convert the data item from a 1D numpy array to a 1D Python list.\n", + " 2. Convert the prediction request to a serialized Google protobuf (`json_format.ParseDict()`)\n", + "\n", + "Each instance in the prediction request is a dictionary entry of the form:\n", + "\n", + " {input_name: content}\n", + "\n", + "- `input_name`: the name of the input layer of the underlying model.\n", + "- `content`: The data item as a 1D Python list.\n", + "\n", + "Since the `predict()` service can take multiple data items (instances), you will send your single data item as a list of one data item. As a final step, you package the instances list into Google's protobuf format -- which is what we pass to the `predict()` service.\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding image in the request. You will see in the output for each prediction:\n", + "\n", + "- `predictions` -- the predicated binary sentiment between 0 (negative) and 1 (positive)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "send_prediction_request:text" + }, + "outputs": [], + "source": [ + "def predict_data(data, endpoint, parameters_dict):\n", + " parameters = json_format.ParseDict(parameters_dict, Value())\n", + "\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [{serving_input: data.tolist()}]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", prediction)\n", + "\n", + "\n", + "predict_data(test_item, endpoint_id, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_custom_text_binary_classification_online_pipeline.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_custom_text_binary_classification_online_tfserving.ipynb b/notebooks/community/gapic/custom/showcase_custom_text_binary_classification_online_tfserving.ipynb new file mode 100644 index 000000000..587a794f3 --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_custom_text_binary_classification_online_tfserving.ipynb @@ -0,0 +1,2169 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Custom training text binary classification model using TF Serving container for online prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom,training,tfserving" + }, + "source": [ + "## Overview\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to train and deploy a custom text binary classification model for online prediction, using TF Serving container for prediction." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:custom,imdb,tbn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [IMDB Movie Reviews](https://www.tensorflow.org/datasets/catalog/imdb_reviews) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts whether a review is positive or negative in sentiment." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,training,online_prediction,tfserving" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial you create a custom model from a Python script in a Google prebuilt Docker container using the Vertex client library, and then do a prediction on the deployed model using TF Serving container. You can alternatively create custom models using `gcloud` command-line tool or online using Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex custom job for training a model.\n", + "- Create a `TrainingPipeline` resource.\n", + "- Train a TensorFlow model.\n", + "- Retrieve and load the model artifacts.\n", + "- View the model evaluation.\n", + "- Upload the model as a Vertex `Model` resource.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource with a TF Serving container.\n", + "- Make a prediction.\n", + "- Undeploy the `Model` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_docker" + }, + "source": [ + "### Install Docker (Colab or Local)\n", + "\n", + "If you are using Google Cloud Notebook, Docker is already installed. Skip these steps.\n", + "\n", + "By default, Docker is not installed on Colab. If you're running colab, then you need to do the following to install Docker." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_docker" + }, + "outputs": [], + "source": [ + "if \"google.colab\" in sys.modules:\n", + " ! sudo apt update\n", + " ! sudo apt install apt-transport-https ca-certificates curl software-properties-common\n", + " ! curl -fsSL https://download.docker.com/linux/ubuntu/gpg | sudo apt-key add -\n", + " ! sudo add-apt-repository \"deb [arch=amd64] https://download.docker.com/linux/ubuntu bionic stable\"\n", + " ! sudo apt update\n", + " ! sudo apt install docker-ce" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "start_docker_service" + }, + "source": [ + "#### Start Docker service\n", + "\n", + "Start the `docker` service." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "start_docker_service" + }, + "outputs": [], + "source": [ + "if \"google.colab\" in sys.modules:\n", + " ! sudo service docker start" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:training,cpu,prediction,cpu" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for training and prediction.\n", + "\n", + "Set the variables `TRAIN_GPU/TRAIN_NGPU` and `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU.\n", + "\n", + "*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3 -- which is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:training,cpu,prediction,cpu" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n", + " TRAIN_GPU, TRAIN_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n", + " )\n", + "else:\n", + " TRAIN_GPU, TRAIN_NGPU = (None, None)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (None, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:training" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for training.\n", + "\n", + "- Set the variable `TF` to the TensorFlow version of the container image. For example, `2-1` would be version 2.1, and `1-15` would be version 1.15. The following list shows some of the pre-built images available:\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-3:latest`\n", + " - TensorFlow 2.4\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-4:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/training/xgboost-cpu.1-1`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/training/scikit-learn-cpu.0-23:latest`\n", + " - Pytorch\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-5:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-6:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-7:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for training](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:training" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + "else:\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "TRAIN_IMAGE = \"gcr.io/cloud-aiplatform/training/{}:latest\".format(TRAIN_VERSION)\n", + "\n", + "print(\"Training:\", TRAIN_IMAGE, TRAIN_GPU, TRAIN_NGPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:tfserving" + }, + "source": [ + "#### Container (Docker) image for prediction\n", + "\n", + "Next, you set the TF Serving Docker container image for prediction.\n", + "\n", + " 1. Pull the corresponding CPU or GPU Docker image for TF Serving from Docker Hub.\n", + " 2. Create a tag for registering the image with Cloud Container Registry (gcr.io)\n", + " 3. Register the image with Cloud Container Registry.\n", + "\n", + "For more details on using TF Serving container, see [TF Serving](https://www.tensorflow.org/tfx/serving/docker)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:tfserving" + }, + "outputs": [], + "source": [ + "if DEPLOY_GPU:\n", + " ! docker pull tensorflow/serving:latest-gpu\n", + " DEPLOY_IMAGE = \"gcr.io/\" + PROJECT_ID + \"/tf_serving:gpu\"\n", + "else:\n", + " ! docker pull tensorflow/serving:latest\n", + " DEPLOY_IMAGE = \"gcr.io/\" + PROJECT_ID + \"/tf_serving\"\n", + "\n", + "! docker tag tensorflow/serving $DEPLOY_IMAGE\n", + "! docker push $DEPLOY_IMAGE\n", + "\n", + "print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU, DEPLOY_NGPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:training,prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for training and prediction.\n", + "\n", + "- Set the variables `TRAIN_COMPUTE` and `DEPLOY_COMPUTE` to configure the compute resources for the VMs you will use for for training and prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: The following is not supported for training:*\n", + "\n", + " - `standard`: 2 vCPUs\n", + " - `highcpu`: 2, 4 and 8 vCPUs\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_TRAIN_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Train machine type\", TRAIN_COMPUTE)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own custom model and training for IMDB Movie Reviews." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:custom" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Model Service for `Model` resources.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:custom" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"job\"] = create_job_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_model" + }, + "source": [ + "## Train a model\n", + "\n", + "There are two ways you can train a custom model using a container image:\n", + "\n", + "- **Use a Google Cloud prebuilt container**. If you use a prebuilt container, you will additionally specify a Python package to install into the container image. This Python package contains your code for training a custom model.\n", + "\n", + "- **Use your own custom container image**. If you use your own container, the container needs to contain your code for training a custom model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_specification:prebuilt_container" + }, + "source": [ + "## Prepare your custom job specification\n", + "\n", + "Now that your clients are ready, your first step is to create a Job Specification for your custom training job. The job specification will consist of the following:\n", + "\n", + "- `worker_pool_spec` : The specification of the type of machine(s) you will use for training and how many (single or distributed)\n", + "- `python_package_spec` : The specification of the Python package to be installed with the pre-built container." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "source": [ + "### Prepare your machine specification\n", + "\n", + "Now define the machine specification for your custom training job. This tells Vertex what type of machine instance to provision for the training.\n", + " - `machine_type`: The type of GCP instance to provision -- e.g., n1-standard-8.\n", + " - `accelerator_type`: The type, if any, of hardware accelerator. In this tutorial if you previously set the variable `TRAIN_GPU != None`, you are using a GPU; otherwise you will use a CPU.\n", + " - `accelerator_count`: The number of accelerators." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "outputs": [], + "source": [ + "if TRAIN_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": TRAIN_COMPUTE,\n", + " \"accelerator_type\": TRAIN_GPU,\n", + " \"accelerator_count\": TRAIN_NGPU,\n", + " }\n", + "else:\n", + " machine_spec = {\"machine_type\": TRAIN_COMPUTE, \"accelerator_count\": 0}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "source": [ + "### Prepare your disk specification\n", + "\n", + "(optional) Now define the disk specification for your custom training job. This tells Vertex what type and size of disk to provision in each machine instance for the training.\n", + "\n", + " - `boot_disk_type`: Either SSD or Standard. SSD is faster, and Standard is less expensive. Defaults to SSD.\n", + " - `boot_disk_size_gb`: Size of disk in GB." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "outputs": [], + "source": [ + "DISK_TYPE = \"pd-ssd\" # [ pd-ssd, pd-standard]\n", + "DISK_SIZE = 200 # GB\n", + "\n", + "disk_spec = {\"boot_disk_type\": DISK_TYPE, \"boot_disk_size_gb\": DISK_SIZE}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container,tfserving" + }, + "source": [ + "### Define the worker pool specification\n", + "\n", + "Next, you define the worker pool specification for your custom training job. The worker pool specification will consist of the following:\n", + "\n", + "- `replica_count`: The number of instances to provision of this machine type.\n", + "- `machine_spec`: The hardware specification.\n", + "- `disk_spec` : (optional) The disk storage specification.\n", + "\n", + "- `python_package`: The Python training package to install on the VM instance(s) and which Python module to invoke, along with command line arguments for the Python module.\n", + "\n", + "Let's dive deeper now into the python package specification:\n", + "\n", + "-`executor_image_spec`: This is the docker image which is configured for your custom training job.\n", + "\n", + "-`package_uris`: This is a list of the locations (URIs) of your python training packages to install on the provisioned instance. The locations need to be in a Cloud Storage bucket. These can be either individual python files or a zip (archive) of an entire package. In the later case, the job service will unzip (unarchive) the contents into the docker image.\n", + "\n", + "-`python_module`: The Python module (script) to invoke for running the custom training job. In this example, you will be invoking `trainer.task.py` -- note that it was not neccessary to append the `.py` suffix.\n", + "\n", + "-`args`: The command line arguments to pass to the corresponding Pythom module. In this example, you will be setting:\n", + " - `\"--model-dir=\" + MODEL_DIR` : The Cloud Storage location where to store the model artifacts. There are two ways to tell the training script where to save the model artifacts:\n", + " - direct: You pass the Cloud Storage location as a command line argument to your training script (set variable `DIRECT = True`), or\n", + " - indirect: The service passes the Cloud Storage location as the environment variable `AIP_MODEL_DIR` to your training script (set variable `DIRECT = False`). In this case, you tell the service the model artifact location in the job specification.\n", + " - `\"--epochs=\" + EPOCHS`: The number of epochs for training.\n", + " - `\"--steps=\" + STEPS`: The number of steps (batches) per epoch.\n", + " - `\"--distribute=\" + TRAIN_STRATEGY\"` : The training distribution strategy to use for single or distributed training.\n", + " - `\"single\"`: single device.\n", + " - `\"mirror\"`: all GPU devices on a single compute instance.\n", + " - `\"multi\"`: all GPU devices on all compute instances.\n", + "\n", + "*Note*: For TF Serving, the `MODEL_DIR` must end in a subfolder that is a number, e.g., 1." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container,tfserving" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_job_\" + TIMESTAMP\n", + "MODEL_DIR = \"{}/{}/1\".format(BUCKET_NAME, JOB_NAME)\n", + "\n", + "if not TRAIN_NGPU or TRAIN_NGPU < 2:\n", + " TRAIN_STRATEGY = \"single\"\n", + "else:\n", + " TRAIN_STRATEGY = \"mirror\"\n", + "\n", + "EPOCHS = 20\n", + "STEPS = 100\n", + "\n", + "DIRECT = True\n", + "if DIRECT:\n", + " CMDARGS = [\n", + " \"--model-dir=\" + MODEL_DIR,\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "else:\n", + " CMDARGS = [\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "\n", + "worker_pool_spec = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": machine_spec,\n", + " \"disk_spec\": disk_spec,\n", + " \"python_package_spec\": {\n", + " \"executor_image_uri\": TRAIN_IMAGE,\n", + " \"package_uris\": [BUCKET_NAME + \"/trainer_imdb.tar.gz\"],\n", + " \"python_module\": \"trainer.task\",\n", + " \"args\": CMDARGS,\n", + " },\n", + " }\n", + "]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "assemble_custom_job_specification" + }, + "source": [ + "### Assemble a job specification\n", + "\n", + "Now assemble the complete description for the custom job specification:\n", + "\n", + "- `display_name`: The human readable name you assign to this custom job.\n", + "- `job_spec`: The specification for the custom job.\n", + " - `worker_pool_specs`: The specification for the machine VM instances.\n", + " - `base_output_directory`: This tells the service the Cloud Storage location where to save the model artifacts (when variable `DIRECT = False`). The service will then pass the location to the training script as the environment variable `AIP_MODEL_DIR`, and the path will be of the form:\n", + "\n", + " /model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "assemble_custom_job_specification" + }, + "outputs": [], + "source": [ + "if DIRECT:\n", + " job_spec = {\"worker_pool_specs\": worker_pool_spec}\n", + "else:\n", + " job_spec = {\n", + " \"worker_pool_specs\": worker_pool_spec,\n", + " \"base_output_directory\": {\"output_uri_prefix\": MODEL_DIR},\n", + " }\n", + "\n", + "custom_job = {\"display_name\": JOB_NAME, \"job_spec\": job_spec}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "examine_training_package" + }, + "source": [ + "### Examine the training package\n", + "\n", + "#### Package layout\n", + "\n", + "Before you start the training, you will look at how a Python package is assembled for a custom training job. When unarchived, the package contains the following directory/file layout.\n", + "\n", + "- PKG-INFO\n", + "- README.md\n", + "- setup.cfg\n", + "- setup.py\n", + "- trainer\n", + " - \\_\\_init\\_\\_.py\n", + " - task.py\n", + "\n", + "The files `setup.cfg` and `setup.py` are the instructions for installing the package into the operating environment of the Docker image.\n", + "\n", + "The file `trainer/task.py` is the Python script for executing the custom training job. *Note*, when we referred to it in the worker pool specification, we replace the directory slash with a dot (`trainer.task`) and dropped the file suffix (`.py`).\n", + "\n", + "#### Package Assembly\n", + "\n", + "In the following cells, you will assemble the training package." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "examine_training_package" + }, + "outputs": [], + "source": [ + "# Make folder for Python training script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\ntag_build =\\n\\ntag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'tensorflow_datasets==1.3.0',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\nName: IMDB Movie Reviews text binary classification\\n\\nVersion: 0.0.0\\n\\nSummary: Demostration training script\\n\\nHome-page: www.google.com\\n\\nAuthor: Google\\n\\nAuthor-email: aferlitsch@google.com\\n\\nLicense: Public\\n\\nDescription: Demo\\n\\nPlatform: Vertex\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:imdb" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the training script task.py. I won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Gets the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n", + "- Loads IMDB Movie Reviews dataset from TF Datasets (tfds).\n", + "- Builds a simple RNN model using TF.Keras model API.\n", + "- Compiles the model (`compile()`).\n", + "- Sets a training distribution strategy according to the argument `args.distribute`.\n", + "- Trains the model (`fit()`) with epochs specified by `args.epochs`.\n", + "- Saves the trained model (`save(args.model_dir)`) to the specified model directory." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:imdb" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Single, Mirror and Multi-Machine Distributed Training for IMDB\n", + "\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "import argparse\n", + "import os\n", + "import sys\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv('AIP_MODEL_DIR'), type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=1e-4, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=20, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=100, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "print(device_lib.list_local_devices())\n", + "\n", + "# Single Machine, single compute device\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "# Single Machine, multiple compute device\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "# Multiple Machine, multiple compute device\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "# Multi-worker configuration\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "# Preparing dataset\n", + "BUFFER_SIZE = 10000\n", + "BATCH_SIZE = 64\n", + "\n", + "\n", + "def make_datasets():\n", + " dataset, info = tfds.load('imdb_reviews/subwords8k', with_info=True,\n", + " as_supervised=True)\n", + " train_dataset, test_dataset = dataset['train'], dataset['test']\n", + " encoder = info.features['text'].encoder\n", + "\n", + " padded_shapes = ([None],())\n", + " return train_dataset.shuffle(BUFFER_SIZE).padded_batch(BATCH_SIZE, padded_shapes), encoder\n", + "\n", + "\n", + "train_dataset, encoder = make_datasets()\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_rnn_model(encoder):\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Embedding(encoder.vocab_size, 64),\n", + " tf.keras.layers.Bidirectional(tf.keras.layers.LSTM(64)),\n", + " tf.keras.layers.Dense(64, activation='relu'),\n", + " tf.keras.layers.Dense(1, activation='sigmoid')\n", + " ])\n", + " model.compile(loss=tf.keras.losses.BinaryCrossentropy(from_logits=True),\n", + " optimizer=tf.keras.optimizers.Adam(args.lr),\n", + " metrics=['accuracy'])\n", + " return model\n", + "\n", + "\n", + "with strategy.scope():\n", + " # Creation of dataset, and model building/compiling need to be within\n", + " # `strategy.scope()`.\n", + " model = build_and_compile_rnn_model(encoder)\n", + "\n", + "# Train the model\n", + "model.fit(train_dataset, epochs=args.epochs, steps_per_epoch=args.steps)\n", + "model.save(args.model_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tarball_training_script" + }, + "source": [ + "#### Store training script on your Cloud Storage bucket\n", + "\n", + "Next, you package the training folder into a compressed tar ball, and then store it in your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tarball_training_script" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_imdb.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job" + }, + "source": [ + "### Train the model\n", + "\n", + "\n", + "Now start the training of your custom training job on Vertex. Use this helper function `create_custom_job`, which takes the following parameter:\n", + "\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "The helper function calls job client service's `create_custom_job` method, with the following parameters:\n", + "\n", + "-`parent`: The Vertex location path to `Dataset`, `Model` and `Endpoint` resources.\n", + "-`custom_job`: The specification for the custom job.\n", + "\n", + "You will display a handful of the fields returned in `response` object, with the two that are of most interest are:\n", + "\n", + "`response.name`: The Vertex fully qualified identifier assigned to this custom training job. You save this identifier for using in subsequent steps.\n", + "\n", + "`response.state`: The current state of the custom training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job" + }, + "outputs": [], + "source": [ + "def create_custom_job(custom_job):\n", + " response = clients[\"job\"].create_custom_job(parent=PARENT, custom_job=custom_job)\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = create_custom_job(custom_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "job_id:response" + }, + "source": [ + "Now get the unique identifier for the custom job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom job\n", + "job_id = response.name\n", + "# The short numeric ID for the custom job\n", + "job_short_id = job_id.split(\"/\")[-1]\n", + "\n", + "print(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_custom_job" + }, + "source": [ + "### Get information on a custom job\n", + "\n", + "Next, use this helper function `get_custom_job`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "The helper function calls the job client service's`get_custom_job` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the custom job.\n", + "\n", + "If you recall, you got the Vertex fully qualified identifier for the custom job in the `response.name` field when you called the `create_custom_job` method, and saved the identifier in the variable `job_id`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_custom_job" + }, + "outputs": [], + "source": [ + "def get_custom_job(name, silent=False):\n", + " response = clients[\"job\"].get_custom_job(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = get_custom_job(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_training_complete:custom" + }, + "source": [ + "# Deployment\n", + "\n", + "Training the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done training, you can calculate the actual time it took to train the model by subtracting `end_time` from `start_time`. For your model, we will need to know the location of the saved model, which the Python script saved in your local Cloud Storage bucket at `MODEL_DIR + '/saved_model.pb'`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_training_complete:custom" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = get_custom_job(job_id, True)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_path_to_deploy = None\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " if not DIRECT:\n", + " MODEL_DIR = MODEL_DIR + \"/model\"\n", + " model_path_to_deploy = MODEL_DIR\n", + " print(\"Training Time:\", response.update_time - response.create_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(\"model_to_deploy:\", model_path_to_deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "load_saved_model" + }, + "source": [ + "## Load the saved model\n", + "\n", + "Your model is stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Now load it from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n", + "\n", + "To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "load_saved_model" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:text" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now let's find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load the IMDB Movie Review test (holdout) data from `tfds.datasets`, using the method `load()`. This will return the dataset as a tuple of two elements. The first element is the dataset and the second is information on the dataset, which will contain the predefined vocabulary encoder. The encoder will convert words into a numerical embedding, which was pretrained and used in the custom training script.\n", + "\n", + "\n", + "When you trained the model, you needed to set a fix input length for your text. For forward feeding batches, the `padded_batch()` property of the corresponding `tf.dataset` was set to pad each input sequence into the same shape for a batch.\n", + "\n", + "For the test data, you also need to set the `padded_batch()` property accordingly." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "evaluate_custom_model:text,imdb" + }, + "outputs": [], + "source": [ + "import tensorflow_datasets as tfds\n", + "\n", + "dataset, info = tfds.load(\"imdb_reviews/subwords8k\", with_info=True, as_supervised=True)\n", + "test_dataset = dataset[\"test\"]\n", + "encoder = info.features[\"text\"].encoder\n", + "\n", + "BATCH_SIZE = 64\n", + "padded_shapes = ([None], ())\n", + "test_dataset = test_dataset.padded_batch(BATCH_SIZE, padded_shapes)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "perform_evaluation_custom:tfds" + }, + "source": [ + "### Perform the model evaluation\n", + "\n", + "Now evaluate how well the model in the custom job did." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "perform_evaluation_custom:tfds" + }, + "outputs": [], + "source": [ + "model.evaluate(test_dataset)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "how_serving_function_works" + }, + "source": [ + "## Upload the model for serving\n", + "\n", + "Next, you will upload your TF.Keras model from the custom job to Vertex `Model` service, which will create a Vertex `Model` resource for your custom model. During upload, you need to define a serving function to convert data to the format your model expects. If you send encoded data to Vertex, your serving function ensures that the data is decoded on the model server before it is passed as input to your model.\n", + "\n", + "### How does the serving function work\n", + "\n", + "When you send a request to an online prediction server, the request is received by a HTTP server. The HTTP server extracts the prediction request from the HTTP request content body. The extracted prediction request is forwarded to the serving function. For Google pre-built prediction containers, the request content is passed to the serving function as a `tf.string`.\n", + "\n", + "The serving function consists of two parts:\n", + "\n", + "- `preprocessing function`:\n", + " - Converts the input (`tf.string`) to the input shape and data type of the underlying model (dynamic graph).\n", + " - Performs the same preprocessing of the data that was done during training the underlying model -- e.g., normalizing, scaling, etc.\n", + "- `post-processing function`:\n", + " - Converts the model output to format expected by the receiving application -- e.q., compresses the output.\n", + " - Packages the output for the the receiving application -- e.g., add headings, make JSON object, etc.\n", + "\n", + "Both the preprocessing and post-processing functions are converted to static graphs which are fused to the model. The output from the underlying model is passed to the post-processing function. The post-processing function passes the converted/packaged output back to the HTTP server. The HTTP server returns the output as the HTTP response content.\n", + "\n", + "One consideration you need to consider when building serving functions for TF.Keras models is that they run as static graphs. That means, you cannot use TF graph operations that require a dynamic graph. If you do, you will get an error during the compile of the serving function which will indicate that you are using an EagerTensor which is not supported." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_signature" + }, + "source": [ + "## Get the serving function signature\n", + "\n", + "You can get the signatures of your model's input and output layers by reloading the model into memory, and querying it for the signatures corresponding to each layer.\n", + "\n", + "When making a prediction request, you need to route the request to the serving function instead of the model, so you need to know the input layer name of the serving function -- which you will use later when you make a prediction request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_signature" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_path_to_deploy)\n", + "\n", + "serving_input = list(\n", + " loaded.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "print(\"Serving function input:\", serving_input)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "upload_the_model:tfserving" + }, + "source": [ + "### Upload the model\n", + "\n", + "Use this helper function `upload_model` to upload your model, stored in SavedModel format, up to the `Model` service, which will instantiate a Vertex `Model` resource instance for your model. Once you've done that, you can use the `Model` resource instance in the same way as any other Vertex `Model` resource instance, such as deploying to an `Endpoint` resource for serving predictions.\n", + "\n", + "The helper function takes the following parameters:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` service.\n", + "- `image_uri`: The container image for the model deployment.\n", + "- `model_uri`: The Cloud Storage path to our SavedModel artificat. For this tutorial, this is the Cloud Storage location where the `trainer/task.py` saved the model artifacts, which we specified in the variable `MODEL_DIR`.\n", + "\n", + "The helper function calls the `Model` client service's method `upload_model`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for `Dataset`, `Model` and `Endpoint` resources.\n", + "- `model`: The specification for the Vertex `Model` resource instance.\n", + "\n", + "Let's now dive deeper into the Vertex model specification `model`. This is a dictionary object that consists of the following fields:\n", + "\n", + "- `display_name`: A human readable name for the `Model` resource.\n", + "- `metadata_schema_uri`: Since your model was built without an Vertex `Dataset` resource, you will leave this blank (`''`).\n", + "- `artificat_uri`: The Cloud Storage path where the model is stored in SavedModel format.\n", + "- `container_spec`: This is the specification for the Docker container that will be installed on the endpoint, from which the model will serve predictions. In this tutoriak, you use a TF Serving as a custom prediction container. Since this is a custom container, you need to tell Vertex how to start up and communicate with the serving binary:\n", + " - `command`: The serving binary (HTTP Server) to start up.\n", + " - `args`: The arguments to pass to the serving binary. For TF Serving, the required arguments are:\n", + " - `--model_name`: The human readable name to assign to the model.\n", + " - `--model_base_name`: Where to store the model artifacts in the container. The Vertex service sets the variable $(AIP_STORAGE_URI) to where the service installed the model artifacts in the container.\n", + " - `--rest_api_port`: The port to which to send REST based prediction requests. Can either be 8080 or 8501 (default for TF Serving).\n", + " - `--port`: The port to which to send gRPC based prediction requests. Should be 8500 for TF Serving.\n", + "\n", + " - `health_route`: The URL for the service to periodically ping for a response to verify that the serving binary is running. For TF Serving, this will be /v1/models/[model_name].\n", + " - `predict_route`: The URL for the service to route REST-based prediction requests to. For TF Serving, this will be /v1/models/[model_name]:predict.\n", + "\n", + "Uploading a model into a Vertex Model resource returns a long running operation, since it may take a few moments. You call response.result(), which is a synchronous call and will return when the Vertex Model resource is ready.\n", + "\n", + "The helper function returns the Vertex fully qualified identifier for the corresponding Vertex Model instance upload_model_response.model. You will save the identifier for subsequent steps in the variable model_to_deploy_id.\n", + "\n", + "*Note*: You drop the ending number subfolder (e.g., /1) from the model path to upload. The Vertex service will upload the parent folder above the subfolder with the model artifacts -- which is what TensorFlow serving binary expects." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "upload_the_model:tfserving" + }, + "outputs": [], + "source": [ + "IMAGE_URI = DEPLOY_IMAGE\n", + "MODEL_NAME = \"imdb-\" + TIMESTAMP\n", + "\n", + "\n", + "def upload_model(display_name, image_uri, model_uri):\n", + " model = {\n", + " \"display_name\": display_name,\n", + " \"metadata_schema_uri\": \"\",\n", + " \"artifact_uri\": model_uri,\n", + " \"container_spec\": {\n", + " \"image_uri\": image_uri,\n", + " \"command\": [\"/usr/bin/tensorflow_model_server\"],\n", + " \"args\": [\n", + " \"--model_name=\" + MODEL_NAME,\n", + " \"--model_base_path=\" + \"$(AIP_STORAGE_URI)\",\n", + " \"--rest_api_port=8080\",\n", + " \"--port=8500\",\n", + " \"--file_system_poll_wait_seconds=31540000\",\n", + " ],\n", + " \"health_route\": \"/v1/models/\" + MODEL_NAME,\n", + " \"predict_route\": \"/v1/models/\" + MODEL_NAME + \":predict\",\n", + " },\n", + " }\n", + " response = clients[\"model\"].upload_model(parent=PARENT, model=model)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " upload_model_response = response.result(timeout=180)\n", + " print(\"upload_model_response\")\n", + " print(\" model:\", upload_model_response.model)\n", + " return upload_model_response.model\n", + "\n", + "\n", + "model_to_deploy_id = upload_model(\n", + " \"imdb-\" + TIMESTAMP, IMAGE_URI, model_path_to_deploy[:-2]\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_model" + }, + "source": [ + "### Get `Model` resource information\n", + "\n", + "Now let's get the model information for just your model. Use this helper function `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource.\n", + "\n", + "This helper function calls the Vertex `Model` client service's method `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_model" + }, + "outputs": [], + "source": [ + "def get_model(name):\n", + " response = clients[\"model\"].get_model(name=name)\n", + " print(response)\n", + "\n", + "\n", + "get_model(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:custom" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex custom `Model` resource. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"imdb_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:dedicated" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `dedicated_resources`: This refers to how many compute instances (replicas) that are scaled for serving prediction requests.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `min_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:dedicated" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"imdb_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " \"machine_spec\": machine_spec,\n", + " },\n", + " \"disable_container_logging\": False,\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "prepare_test_item:text" + }, + "source": [ + "### Prepare the request content\n", + "\n", + "Since the dataset is a `tf.dataset`, which acts as a generator, we must use it as an iterator to access the data items in the test data. We do the following to get a single data item from the test data:\n", + "\n", + "- Set the property for the number of batches to draw per iteration to one using the method `take(1)`.\n", + "- Iterate once through the test data -- i.e., we do a break within the for loop.\n", + "- In the single iteration, we save the data item which is in the form of a tuple.\n", + "- The data item will be the first element of the tuple, which you then will convert from an tensor to a numpy array -- `data[0].numpy()`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "prepare_test_item:text" + }, + "outputs": [], + "source": [ + "import tensorflow_datasets as tfds\n", + "\n", + "dataset, info = tfds.load(\"imdb_reviews/subwords8k\", with_info=True, as_supervised=True)\n", + "test_dataset = dataset[\"test\"]\n", + "\n", + "test_dataset.take(1)\n", + "for data in test_dataset:\n", + " print(data)\n", + " break\n", + "\n", + "test_item = data[0].numpy()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "send_prediction_request:text" + }, + "source": [ + "### Send the prediction request\n", + "\n", + "Ok, now you have a test data item. Use this helper function `predict_data`, which takes the following parameters:\n", + "\n", + "- `data`: The test data item is a 64 padded numpy 1D array.\n", + "- `endpoint`: The Vertex AI fully qualified identifier for the endpoint where the model was deployed.\n", + "- `parameters_dict`: Additional parameters for serving.\n", + "\n", + "This function uses the prediction client service and calls the `predict` method with the following parameters:\n", + "\n", + "- `endpoint`: The Vertex AI fully qualified identifier for the endpoint where the model was deployed.\n", + "- `instances`: A list of instances (data items) to predict.\n", + "- `parameters`: Additional parameters for serving.\n", + "\n", + "To pass the test data to the prediction service, you must package it for transmission to the serving binary as follows:\n", + "\n", + " 1. Convert the data item from a 1D numpy array to a 1D Python list.\n", + " 2. Convert the prediction request to a serialized Google protobuf (`json_format.ParseDict()`)\n", + "\n", + "Each instance in the prediction request is a dictionary entry of the form:\n", + "\n", + " {input_name: content}\n", + "\n", + "- `input_name`: the name of the input layer of the underlying model.\n", + "- `content`: The data item as a 1D Python list.\n", + "\n", + "Since the `predict()` service can take multiple data items (instances), you will send your single data item as a list of one data item. As a final step, you package the instances list into Google's protobuf format -- which is what we pass to the `predict()` service.\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding image in the request. You will see in the output for each prediction:\n", + "\n", + "- `predictions` -- the predicated binary sentiment between 0 (negative) and 1 (positive)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "send_prediction_request:text" + }, + "outputs": [], + "source": [ + "def predict_data(data, endpoint, parameters_dict):\n", + " parameters = json_format.ParseDict(parameters_dict, Value())\n", + "\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [{serving_input: data.tolist()}]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", prediction)\n", + "\n", + "\n", + "predict_data(test_item, endpoint_id, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_custom_text_binary_classification_online_tfserving.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_custom_text_classification_online_exported_ds.ipynb b/notebooks/community/gapic/custom/showcase_custom_text_classification_online_exported_ds.ipynb new file mode 100644 index 000000000..910430a28 --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_custom_text_classification_online_exported_ds.ipynb @@ -0,0 +1,1305 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Custom training text classification model for online prediction using exported dataset\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom,exported_ds" + }, + "source": [ + "## Overview\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to train and deploy a custom text classification model for online prediction, using an exported `Dataset` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:happydb,tcn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Happy Moments dataset](https://www.kaggle.com/ritresearch/happydb) from [Kaggle Datasets](https://www.kaggle.com/ritresearch/happydb). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,exported_ds,online_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you learn how to create a custom model using an exported `Dataset` resource from a Python script in a Docker container using the Vertex client library, and then do a prediction on the deployed model. You can alternatively create models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Export the `Dataset` resource's manifest.\n", + "- Create a Vertex custom job for training a model.\n", + "- Import the exported dataset manifest.\n", + "- Train the model.\n", + "- Retrieve and load the model artifacts.\n", + "- View the model evaluation.\n", + "- Upload the model as a Vertex `Model` resource.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "labeling_constants:tcn" + }, + "outputs": [], + "source": [ + "# Text Dataset type\n", + "DATA_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/metadata/text_1.0.0.yaml\"\n", + "# Text Labeling type\n", + "LABEL_SCHEMA = \"gs://google-cloud-aiplatform/schema/dataset/ioformat/text_classification_single_label_io_format_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for training and prediction.\n", + "\n", + "Set the variables `TRAIN_GPU/TRAIN_NGPU` and `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU.\n", + "\n", + "*Note*: TF releases before 2.3 for GPU support will fail to load the custom model in this tutorial. It is a known issue and fixed in TF 2.3 -- which is caused by static graph ops that are generated in the serving function. If you encounter this issue on your own custom models, use a container image for TF 2.3 with GPU support." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:training,prediction,cpu" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n", + " TRAIN_GPU, TRAIN_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n", + " )\n", + "else:\n", + " TRAIN_GPU, TRAIN_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (None, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:training,prediction" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for training and prediction\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-3:latest`\n", + " - TensorFlow 2.4\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-4:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/training/xgboost-cpu.1-1`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/training/scikit-learn-cpu.0-23:latest`\n", + " - Pytorch\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-5:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-6:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-7:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for training](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers).\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-3:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-90:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-82:latest`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-23:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-22:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-20:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for prediction](https://cloud.google.com/vertex-ai/docs/predictions/pre-built-containers)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf2-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf2-cpu.{}\".format(TF)\n", + "else:\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "TRAIN_IMAGE = \"gcr.io/cloud-aiplatform/training/{}:latest\".format(TRAIN_VERSION)\n", + "DEPLOY_IMAGE = \"gcr.io/cloud-aiplatform/prediction/{}:latest\".format(DEPLOY_VERSION)\n", + "\n", + "print(\"Training:\", TRAIN_IMAGE, TRAIN_GPU, TRAIN_NGPU)\n", + "print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU, DEPLOY_NGPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:training,prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for training and prediction.\n", + "\n", + "- Set the variables `TRAIN_COMPUTE` and `DEPLOY_COMPUTE` to configure the compute resources for the VMs you will use for for training and prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: The following is not supported for training:*\n", + "\n", + " - `standard`: 2 vCPUs\n", + " - `highcpu`: 2, 4 and 8 vCPUs\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:training,prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_TRAIN_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Train machine type\", TRAIN_COMPUTE)\n", + "\n", + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own custom model and training for Happy Moments." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:custom,exported_ds" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Dataset Service for `Dataset` resources.\n", + "- Model Service for `Model` resources.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:custom,exported_ds" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"job\"] = create_job_client()\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_aip_dataset" + }, + "source": [ + "## Dataset\n", + "\n", + "Now that your clients are ready, your first step in training a model is to create a managed dataset instance, and then upload your labeled data to it.\n", + "\n", + "### Create `Dataset` resource instance\n", + "\n", + "Use the helper function `create_dataset` to create the instance of a `Dataset` resource. This function does the following:\n", + "\n", + "1. Uses the dataset client service.\n", + "2. Creates an Vertex `Dataset` resource (`aip.Dataset`), with the following parameters:\n", + " - `display_name`: The human-readable name you choose to give it.\n", + " - `metadata_schema_uri`: The schema for the dataset type.\n", + "3. Calls the client dataset service method `create_dataset`, with the following parameters:\n", + " - `parent`: The Vertex location root path for your `Database`, `Model` and `Endpoint` resources.\n", + " - `dataset`: The Vertex dataset object instance you created.\n", + "4. The method returns an `operation` object.\n", + "\n", + "An `operation` object is how Vertex handles asynchronous calls for long running operations. While this step usually goes fast, when you first use it in your project, there is a longer delay due to provisioning.\n", + "\n", + "You can use the `operation` object to get status on the operation (e.g., create `Dataset` resource) or to cancel the operation, by invoking an operation method:\n", + "\n", + "| Method | Description |\n", + "| ----------- | ----------- |\n", + "| result() | Waits for the operation to complete and returns a result object in JSON format. |\n", + "| running() | Returns True/False on whether the operation is still running. |\n", + "| done() | Returns True/False on whether the operation is completed. |\n", + "| canceled() | Returns True/False on whether the operation was canceled. |\n", + "| cancel() | Cancels the operation (this may take up to 30 seconds). |" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_aip_dataset" + }, + "outputs": [], + "source": [ + "TIMEOUT = 90\n", + "\n", + "\n", + "def create_dataset(name, schema, labels=None, timeout=TIMEOUT):\n", + " start_time = time.time()\n", + " try:\n", + " dataset = aip.Dataset(\n", + " display_name=name, metadata_schema_uri=schema, labels=labels\n", + " )\n", + "\n", + " operation = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)\n", + " print(\"Long running operation:\", operation.operation.name)\n", + " result = operation.result(timeout=TIMEOUT)\n", + " print(\"time:\", time.time() - start_time)\n", + " print(\"response\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" metadata_schema_uri:\", result.metadata_schema_uri)\n", + " print(\" metadata:\", dict(result.metadata))\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " print(\" etag:\", result.etag)\n", + " print(\" labels:\", dict(result.labels))\n", + " return result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "result = create_dataset(\"happydb-\" + TIMESTAMP, DATA_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset_id:result" + }, + "source": [ + "Now save the unique dataset identifier for the `Dataset` resource instance you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:text,u_dataset" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for text has a couple of requirements for your text data.\n", + "\n", + "- Text examples must be stored in a CSV or JSONL file." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:tcn,u_dataset,csv" + }, + "source": [ + "#### CSV\n", + "\n", + "For text classification, the CSV file has a few requirements:\n", + "\n", + "- No heading.\n", + "- First column is the text example or Cloud Storage path to text file (.txt suffix).\n", + "- Second column the label." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:happydb,csv,tcn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://cloud-ml-data/NL-classification/happiness.csv\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:csv" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Happy Moments dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:csv" + }, + "outputs": [], + "source": [ + "if \"IMPORT_FILES\" in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_data" + }, + "source": [ + "### Import data\n", + "\n", + "Now, import the data into your Vertex Dataset resource. Use this helper function `import_data` to import the data. The function does the following:\n", + "\n", + "- Uses the `Dataset` client.\n", + "- Calls the client method `import_data`, with the following parameters:\n", + " - `name`: The human readable name you give to the `Dataset` resource (e.g., happydb).\n", + " - `import_configs`: The import configuration.\n", + "\n", + "- `import_configs`: A Python list containing a dictionary, with the key/value entries:\n", + " - `gcs_sources`: A list of URIs to the paths of the one or more index files.\n", + " - `import_schema_uri`: The schema identifying the labeling type.\n", + "\n", + "The `import_data()` method returns a long running `operation` object. This will take a few minutes to complete. If you are in a live tutorial, this would be a good time to ask questions, or take a personal break." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_data" + }, + "outputs": [], + "source": [ + "def import_data(dataset, gcs_sources, schema):\n", + " config = [{\"gcs_source\": {\"uris\": gcs_sources}, \"import_schema_uri\": schema}]\n", + " print(\"dataset:\", dataset_id)\n", + " start_time = time.time()\n", + " try:\n", + " operation = clients[\"dataset\"].import_data(\n", + " name=dataset_id, import_configs=config\n", + " )\n", + " print(\"Long running operation:\", operation.operation.name)\n", + "\n", + " result = operation.result()\n", + " print(\"result:\", result)\n", + " print(\"time:\", int(time.time() - start_time), \"secs\")\n", + " print(\"error:\", operation.exception())\n", + " print(\"meta :\", operation.metadata)\n", + " print(\n", + " \"after: running:\",\n", + " operation.running(),\n", + " \"done:\",\n", + " operation.done(),\n", + " \"cancelled:\",\n", + " operation.cancelled(),\n", + " )\n", + "\n", + " return operation\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None\n", + "\n", + "\n", + "import_data(dataset_id, [IMPORT_FILE], LABEL_SCHEMA)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "export_dataset" + }, + "source": [ + "### Export dataset index\n", + "\n", + "Next, you will export the dataset index to a JSONL file which will then be used by your custom training job to get the data and corresponding labels for training your Happy Moments model. Use this helper function `export_data` to export the dataset index. The function does the following:\n", + "\n", + "- Uses the dataset client.\n", + "- Calls the client method `export_data`, with the following parameters:\n", + " - `name`: The human readable name you give to the dataset (e.g., happydb).\n", + " - `export_config`: The export configuration.\n", + "- `export_config` A python list containing a dictionary, with the key/value entries:\n", + " - `gcs_destination`: The Cloud Storage bucket to write the JSONL dataset index file to.\n", + "\n", + "The `export_data()` method returns a long running `operation` object. This will take a few minutes to complete. The helper function will return the long running operation and the result of the operation when the export has completed." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "export_dataset" + }, + "outputs": [], + "source": [ + "EXPORT_FILE = BUCKET_NAME + \"/export\"\n", + "\n", + "\n", + "def export_data(dataset_id, gcs_dest):\n", + " config = {\"gcs_destination\": {\"output_uri_prefix\": gcs_dest}}\n", + " start_time = time.time()\n", + " try:\n", + " operation = clients[\"dataset\"].export_data(\n", + " name=dataset_id, export_config=config\n", + " )\n", + " print(\"Long running operation:\", operation.operation.name)\n", + "\n", + " result = operation.result()\n", + " print(\"result:\", result)\n", + " print(\"time:\", int(time.time() - start_time), \"secs\")\n", + " print(\"error:\", operation.exception())\n", + " print(\"meta :\", operation.metadata)\n", + " print(\n", + " \"after: running:\",\n", + " operation.running(),\n", + " \"done:\",\n", + " operation.done(),\n", + " \"cancelled:\",\n", + " operation.cancelled(),\n", + " )\n", + "\n", + " return operation, result\n", + " except Exception as e:\n", + " print(\"exception:\", e)\n", + " return None, None\n", + "\n", + "\n", + "_, result = export_data(dataset_id, EXPORT_FILE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "export_dataset_quick_peak:tcn" + }, + "source": [ + "#### Quick peak at your exported dataset index file\n", + "\n", + "Let's now take a quick peak at the contents of the exported dataset index file. When the `export_data()` completed, the response object was obtained from the `result()` method of the long running operation. The response object contains the property:\n", + "\n", + "- `exported_files`: A list of the paths to the exported dataset index files, which in this case will be one file.\n", + "\n", + "You will get the path to the exported dataset index file (`result.exported_files[0]`) and then display the first ten JSON objects in the file -- i.e., data items.\n", + "\n", + "The JSONL format for each data item is:\n", + "\n", + " { \"textGcsUri\": path_to_the_text_file, \"classificationAnnotation\": { \"displayName\": label } }" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "export_dataset_quick_peak:tcn" + }, + "outputs": [], + "source": [ + "jsonl_index = result.exported_files[0]\n", + "\n", + "! gsutil cat $jsonl_index | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_custom_text_classification_online_exported_ds.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_hyperparmeter_tuning_image_classification.ipynb b/notebooks/community/gapic/custom/showcase_hyperparmeter_tuning_image_classification.ipynb new file mode 100644 index 000000000..e0a871768 --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_hyperparmeter_tuning_image_classification.ipynb @@ -0,0 +1,2081 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Hyperparameter tuning image classification model\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom,hpt" + }, + "source": [ + "## Overview\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to do hyperparameter tuning for a custom image classification model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:custom,cifar10,icn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [CIFAR10 dataset](https://www.tensorflow.org/datasets/catalog/cifar10) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use is built into TensorFlow. The trained model predicts which type of class an image is from ten classes: airplane, automobile, bird, cat, deer, dog, frog, horse, ship, truck." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,hpt" + }, + "source": [ + "### Objective\n", + "\n", + "In this notebook, you learn how to create a hyperparameter tuning job for a custom image classification model from a Python script in a docker container using the Vertex client library. You can alternatively hyperparameter tune models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create an Vertex hyperparameter turning job for training a custom model.\n", + "- Tune the custom model.\n", + "- Evaluate the study results." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:training" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n", + " TRAIN_GPU, TRAIN_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n", + " )\n", + "else:\n", + " TRAIN_GPU, TRAIN_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:training" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for training.\n", + "\n", + "- Set the variable `TF` to the TensorFlow version of the container image. For example, `2-1` would be version 2.1, and `1-15` would be version 1.15. The following list shows some of the pre-built images available:\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-3:latest`\n", + " - TensorFlow 2.4\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-4:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/training/xgboost-cpu.1-1`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/training/scikit-learn-cpu.0-23:latest`\n", + " - Pytorch\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-5:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-6:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-7:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for training](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:training" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + "else:\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "TRAIN_IMAGE = \"gcr.io/cloud-aiplatform/training/{}:latest\".format(TRAIN_VERSION)\n", + "\n", + "print(\"Training:\", TRAIN_IMAGE, TRAIN_GPU, TRAIN_NGPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:training" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for training.\n", + "\n", + "- Set the variable `TRAIN_COMPUTE` to configure the compute resources for the VMs you will use for for training.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: The following is not supported for training:*\n", + "\n", + " - `standard`: 2 vCPUs\n", + " - `highcpu`: 2, 4 and 8 vCPUs\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:training" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_TRAIN_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Train machine type\", TRAIN_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom,hpt" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own hyperparameter tuning and training of a custom image classification." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:custom,hpt" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Model Service for `Model` resources.\n", + "- Job Service for hyperparameter tuning." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:custom,hpt" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"job\"] = create_job_client()\n", + "clients[\"model\"] = create_model_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tune_custom_model:simple" + }, + "source": [ + "## Tuning a model - Hello World\n", + "\n", + "There are two ways you can hyperparameter tune and train a custom model using a container image:\n", + "\n", + "- **Use a Google Cloud prebuilt container**. If you use a prebuilt container, you will additionally specify a Python package to install into the container image. This Python package contains your code for hyperparameter tuning and training a custom model.\n", + "\n", + "- **Use your own custom container image**. If you use your own container, the container needs to contain your code for hyperparameter tuning and training a custom model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_specification:prebuilt_container,hpt" + }, + "source": [ + "## Prepare your hyperparameter tuning job specification\n", + "\n", + "Now that your clients are ready, your first step is to create a Job Specification for your hyperparameter tuning job. The job specification will consist of the following:\n", + "\n", + "- `trial_job_spec`: The specification for the custom job.\n", + " - `worker_pool_spec` : The specification of the type of machine(s) you will use for hyperparameter tuning and how many (single or distributed)\n", + " - `python_package_spec` : The specification of the Python package to be installed with the pre-built container.\n", + "\n", + "- `study_spec`: The specification for what to tune.\n", + " - `parameters`: This is the specification of the hyperparameters that you will tune for the custom training job. It will contain a list of the\n", + " - `metrics`: This is the specification on how to evaluate the result of each tuning trial." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "source": [ + "### Prepare your machine specification\n", + "\n", + "Now define the machine specification for your custom hyperparameter tuning job. This tells Vertex what type of machine instance to provision for the hyperparameter tuning.\n", + " - `machine_type`: The type of GCP instance to provision -- e.g., n1-standard-8.\n", + " - `accelerator_type`: The type, if any, of hardware accelerator. In this tutorial if you previously set the variable `TRAIN_GPU != None`, you are using a GPU; otherwise you will use a CPU.\n", + " - `accelerator_count`: The number of accelerators." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "outputs": [], + "source": [ + "if TRAIN_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": TRAIN_COMPUTE,\n", + " \"accelerator_type\": TRAIN_GPU,\n", + " \"accelerator_count\": TRAIN_NGPU,\n", + " }\n", + "else:\n", + " machine_spec = {\"machine_type\": TRAIN_COMPUTE, \"accelerator_count\": 0}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "source": [ + "### Prepare your disk specification\n", + "\n", + "(optional) Now define the disk specification for your custom hyperparameter tuning job. This tells Vertex what type and size of disk to provision in each machine instance for the hyperparameter tuning.\n", + "\n", + " - `boot_disk_type`: Either SSD or Standard. SSD is faster, and Standard is less expensive. Defaults to SSD.\n", + " - `boot_disk_size_gb`: Size of disk in GB." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "outputs": [], + "source": [ + "DISK_TYPE = \"pd-ssd\" # [ pd-ssd, pd-standard]\n", + "DISK_SIZE = 200 # GB\n", + "\n", + "disk_spec = {\"boot_disk_type\": DISK_TYPE, \"boot_disk_size_gb\": DISK_SIZE}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container" + }, + "source": [ + "### Define the worker pool specification\n", + "\n", + "Next, you define the worker pool specification for your custom hyperparameter tuning job. The worker pool specification will consist of the following:\n", + "\n", + "- `replica_count`: The number of instances to provision of this machine type.\n", + "- `machine_spec`: The hardware specification.\n", + "- `disk_spec` : (optional) The disk storage specification.\n", + "\n", + "- `python_package`: The Python training package to install on the VM instance(s) and which Python module to invoke, along with command line arguments for the Python module.\n", + "\n", + "Let's dive deeper now into the python package specification:\n", + "\n", + "-`executor_image_spec`: This is the docker image which is configured for your custom hyperparameter tuning job.\n", + "\n", + "-`package_uris`: This is a list of the locations (URIs) of your python training packages to install on the provisioned instance. The locations need to be in a Cloud Storage bucket. These can be either individual python files or a zip (archive) of an entire package. In the later case, the job service will unzip (unarchive) the contents into the docker image.\n", + "\n", + "-`python_module`: The Python module (script) to invoke for running the custom hyperparameter tuning job. In this example, you will be invoking `trainer.task.py` -- note that it was not neccessary to append the `.py` suffix.\n", + "\n", + "-`args`: The command line arguments to pass to the corresponding Pythom module. In this example, you will be setting:\n", + " - `\"--model-dir=\" + MODEL_DIR` : The Cloud Storage location where to store the model artifacts. There are two ways to tell the hyperparameter tuning script where to save the model artifacts:\n", + " - direct: You pass the Cloud Storage location as a command line argument to your training script (set variable `DIRECT = True`), or\n", + " - indirect: The service passes the Cloud Storage location as the environment variable `AIP_MODEL_DIR` to your training script (set variable `DIRECT = False`). In this case, you tell the service the model artifact location in the job specification.\n", + " - `\"--epochs=\" + EPOCHS`: The number of epochs for training.\n", + " - `\"--steps=\" + STEPS`: The number of steps (batches) per epoch.\n", + " - `\"--distribute=\" + TRAIN_STRATEGY\"` : The hyperparameter tuning distribution strategy to use for single or distributed hyperparameter tuning.\n", + " - `\"single\"`: single device.\n", + " - `\"mirror\"`: all GPU devices on a single compute instance.\n", + " - `\"multi\"`: all GPU devices on all compute instances." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_job_\" + TIMESTAMP\n", + "MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, JOB_NAME)\n", + "\n", + "if not TRAIN_NGPU or TRAIN_NGPU < 2:\n", + " TRAIN_STRATEGY = \"single\"\n", + "else:\n", + " TRAIN_STRATEGY = \"mirror\"\n", + "\n", + "EPOCHS = 20\n", + "STEPS = 100\n", + "\n", + "DIRECT = True\n", + "if DIRECT:\n", + " CMDARGS = [\n", + " \"--model-dir=\" + MODEL_DIR,\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "else:\n", + " CMDARGS = [\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "\n", + "worker_pool_spec = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": machine_spec,\n", + " \"disk_spec\": disk_spec,\n", + " \"python_package_spec\": {\n", + " \"executor_image_uri\": TRAIN_IMAGE,\n", + " \"package_uris\": [BUCKET_NAME + \"/trainer_cifar10.tar.gz\"],\n", + " \"python_module\": \"trainer.task\",\n", + " \"args\": CMDARGS,\n", + " },\n", + " }\n", + "]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_study_spec:simple" + }, + "source": [ + "### Create a study specification\n", + "\n", + "Let's start with a simple study. You will just use a single parameter -- the *learning rate*. Since its just one parameter, it doesn't make much sense to do a random search. Instead, we will do a grid search over a range of values.\n", + "\n", + "- `metrics`:\n", + " - `metric_id`: In this example, the objective metric to report back is `'val_accuracy'`\n", + " - `goal`: In this example, the hyperparameter tuning service will evaluate trials to maximize the value of the objective metric.\n", + "- `parameters`: The specification for the hyperparameters to tune.\n", + " - `parameter_id`: The name of the hyperparameter that will be passed to the Python package as a command line argument.\n", + " - `scale_type`: The scale type determines the resolution the hyperparameter tuning service uses when searching over the search space.\n", + " - `UNIT_LINEAR_SCALE`: Uses a resolution that is the same everywhere in the search space.\n", + " - `UNIT_LOG_SCALE`: Values close to the bottom of the search space are further away.\n", + " - `UNIT_REVERSE_LOG_SCALE`: Values close to the top of the search space are further away.\n", + " - **search space**: This is where you will specify the search space of values for the hyperparameter to select for tuning.\n", + " - `integer_value_spec`: Specifies an integer range of values between a `min_value` and `max_value`.\n", + " - `double_value_spec`: Specifies a continuous range of values between a `min_value` and `max_value`.\n", + " - `discrete_value_spec`: Specifies a list of values.\n", + "- `algorithm`: The search method for selecting hyperparameter values per trial:\n", + " - `GRID_SEARCH`: Combinatorically search -- which is used in this example.\n", + " - `RANDOM_SEARCH`: Random search.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_study_spec:simple" + }, + "outputs": [], + "source": [ + "study_spec = {\n", + " \"metrics\": [\n", + " {\n", + " \"metric_id\": \"val_accuracy\",\n", + " \"goal\": aip.StudySpec.MetricSpec.GoalType.MAXIMIZE,\n", + " }\n", + " ],\n", + " \"parameters\": [\n", + " {\n", + " \"parameter_id\": \"lr\",\n", + " \"discrete_value_spec\": {\"values\": [0.001, 0.01, 0.1]},\n", + " \"scale_type\": aip.StudySpec.ParameterSpec.ScaleType.UNIT_LINEAR_SCALE,\n", + " }\n", + " ],\n", + " \"algorithm\": aip.StudySpec.Algorithm.GRID_SEARCH,\n", + "}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "assemble_custom_hpt_job_specification" + }, + "source": [ + "### Assemble a hyperparameter tuning job specification\n", + "\n", + "Now assemble the complete description for the custom hyperparameter tuning specification:\n", + "\n", + "- `display_name`: The human readable name you assign to this custom hyperparameter tuning job.\n", + "- `trial_job_spec`: The specification for the custom hyperparameter tuning job.\n", + "- `study_spec`: The specification for what to tune.\n", + "- `max_trial_count`: The maximum number of tuning trials.\n", + "- `parallel_trial_count`: How many trials to try in parallel; otherwise, they are done sequentially." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "assemble_custom_hpt_job_specification" + }, + "outputs": [], + "source": [ + "hpt_job = {\n", + " \"display_name\": JOB_NAME,\n", + " \"trial_job_spec\": {\"worker_pool_specs\": worker_pool_spec},\n", + " \"study_spec\": study_spec,\n", + " \"max_trial_count\": 6,\n", + " \"parallel_trial_count\": 1,\n", + "}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "examine_training_package" + }, + "source": [ + "### Examine the hyperparameter tuning package\n", + "\n", + "#### Package layout\n", + "\n", + "Before you start the hyperparameter tuning, you will look at how a Python package is assembled for a custom hyperparameter tuning job. When unarchived, the package contains the following directory/file layout.\n", + "\n", + "- PKG-INFO\n", + "- README.md\n", + "- setup.cfg\n", + "- setup.py\n", + "- trainer\n", + " - \\_\\_init\\_\\_.py\n", + " - task.py\n", + "\n", + "The files `setup.cfg` and `setup.py` are the instructions for installing the package into the operating environment of the Docker image.\n", + "\n", + "The file `trainer/task.py` is the Python script for executing the custom hyperparameter tuning job. *Note*, when we referred to it in the worker pool specification, we replace the directory slash with a dot (`trainer.task`) and dropped the file suffix (`.py`).\n", + "\n", + "#### Package Assembly\n", + "\n", + "In the following cells, you will assemble the training package." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "examine_training_package" + }, + "outputs": [], + "source": [ + "# Make folder for Python hyperparameter tuning script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\ntag_build =\\n\\ntag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'tensorflow_datasets==1.3.0',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\nName: CIFAR10 image classification\\n\\nVersion: 0.0.0\\n\\nSummary: Demostration hyperparameter tuning script\\n\\nHome-page: www.google.com\\n\\nAuthor: Google\\n\\nAuthor-email: aferlitsch@google.com\\n\\nLicense: Public\\n\\nDescription: Demo\\n\\nPlatform: Vertex\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:hpt,simple" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the hyperparameter tuning script task.py. I won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Passes the hyperparameter values for a trial as a command line argument (`parser.add_argument('--lr',...)`)\n", + "- Mimics a training loop, where on each loop (epoch) the variable `accuracy` is set to the loop iteration * the learning rate.\n", + "- Reports back the objective metric `accuracy` back to the hyperparameter tuning service using `report_hyperparameter_tuning_metric()`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:hpt,simple" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# HP Tuning hello world example\n", + "\n", + "from __future__ import absolute_import, division, print_function, unicode_literals\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "from hypertune import HyperTune\n", + "import argparse\n", + "import os\n", + "import sys\n", + "import time\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.001, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=10, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=200, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--model-dir',\n", + " dest='model_dir',\n", + " default='/tmp/saved_model',\n", + " type=str,\n", + " help='Model dir.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "print(device_lib.list_local_devices())\n", + "\n", + "# Instantiate the HyperTune reporting object\n", + "hpt = HyperTune()\n", + "\n", + "for epoch in range(1, args.epochs+1):\n", + " # mimic metric result at the end of an epoch\n", + " acc = args.lr * epoch\n", + " # save the metric result to communicate back to the HPT service\n", + " hpt.report_hyperparameter_tuning_metric(\n", + " hyperparameter_metric_tag='val_accuracy',\n", + " metric_value=acc,\n", + " global_step=epoch)\n", + " print('epoch: {}, accuracy: {}'.format(epoch, acc))\n", + " time.sleep(1)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tarball_training_script" + }, + "source": [ + "#### Store hyperparameter tuning script on your Cloud Storage bucket\n", + "\n", + "Next, you package the hyperparameter tuning folder into a compressed tar ball, and then store it in your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tarball_training_script" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_cifar10.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "report_hypertune" + }, + "source": [ + "#### Reporting back the result of the trial using hypertune\n", + "\n", + "For each trial, your Python script needs to report back to the hyperparameter tuning service the objective metric for which you specified as the criteria for evaluating the trial.\n", + "\n", + "For this example, you will specify in the study specification that the objective metric will be reported back as `val_accuracy`.\n", + "\n", + "You report back the value of the objective metric using `HyperTune`. This Python module is used to communicate key/value pairs to the hyperparameter tuning service. To setup this reporting in your Python package, you will add code for the following three steps:\n", + "\n", + "1. Import the HyperTune module: `from hypertune import HyperTune()`.\n", + "2. At the end of every epoch, write the current value of the objective function to the log as a key/value pair using `hpt.report_hyperparameter_tuning_metric()`. In this example, the parameters are:\n", + " - `hyperparameter_metric_tag`: The name of the objective metric to report back. The name must be identical to the name specified in the study specification.\n", + " - `metric_value`: The value of the objective metric to report back to the hyperparameter service.\n", + " - `global_step`: The epoch iteration, starting at 0." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tune_custom_job" + }, + "source": [ + "## Hyperparameter Tune the model\n", + "\n", + "Now start the hyperparameter tuning of your custom model on Vertex. Use this helper function `create_hyperparameter_tuning_job`, which takes the following parameter:\n", + "\n", + "-`hpt_job`: The specification for the hyperparameter tuning job.\n", + "\n", + "The helper function calls job client service's `create_hyperparameter_tuning_job` method, with the following parameters:\n", + "\n", + "-`parent`: The Vertex location path to `Dataset`, `Model` and `Endpoint` resources.\n", + "-`hyperparameter_tuning_job`: The specification for the hyperparameter tuning job.\n", + "\n", + "You will display a handful of the fields returned in `response` object, with the two that are of most interest are:\n", + "\n", + "`response.name`: The Vertex fully qualified identifier assigned to this custom hyperparameter tuning job. You save this identifier for using in subsequent steps.\n", + "\n", + "`response.state`: The current state of the custom hyperparameter tuning job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tune_custom_job" + }, + "outputs": [], + "source": [ + "def create_hyperparameter_tuning_job(hpt_job):\n", + " response = clients[\"job\"].create_hyperparameter_tuning_job(\n", + " parent=PARENT, hyperparameter_tuning_job=hpt_job\n", + " )\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = create_hyperparameter_tuning_job(hpt_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "hpt_job_id:response" + }, + "source": [ + "Now get the unique identifier for the hyperparameter tuning job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "hpt_job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the hyperparameter tuning job\n", + "hpt_job_id = response.name\n", + "# The short numeric ID for the hyperparameter tuning job\n", + "hpt_job_short_id = hpt_job_id.split(\"/\")[-1]\n", + "\n", + "print(hpt_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_hpt_job" + }, + "source": [ + "### Get information on a hyperparameter tuning job\n", + "\n", + "Next, use this helper function `get_hyperparameter_tuning_job`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the hyperparameter tuning job.\n", + "\n", + "The helper function calls the job client service's `get_hyperparameter_tuning_job` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the hyperparameter tuning job.\n", + "\n", + "If you recall, you got the Vertex fully qualified identifier for the hyperparameter tuning job in the `response.name` field when you called the `create_hyperparameter_tuning_job` method, and saved the identifier in the variable `hpt_job_id`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_hpt_job" + }, + "outputs": [], + "source": [ + "def get_hyperparameter_tuning_job(name, silent=False):\n", + " response = clients[\"job\"].get_hyperparameter_tuning_job(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = get_hyperparameter_tuning_job(hpt_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_tuning_complete" + }, + "source": [ + "## Wait for tuning to complete\n", + "\n", + "Hyperparameter tuning the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done tuning, you can calculate the actual time it took to tune the model by subtracting `end_time` from `start_time`.\n", + "\n", + "For your model, we will need to know the location of the saved models for each trial, which the Python script saved in your local Cloud Storage bucket at `MODEL_DIR + '//saved_model.pb'`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_tuning_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " job_response = get_hyperparameter_tuning_job(hpt_job_id, True)\n", + " if job_response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"Study trials have not completed:\", job_response.state)\n", + " if job_response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " if not DIRECT:\n", + " MODEL_DIR = MODEL_DIR + \"/model\"\n", + " print(\"Study trials have completed\")\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "review_study_results" + }, + "source": [ + "### Review the results of the study\n", + "\n", + "Now review the results of trials." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "review_study_results" + }, + "outputs": [], + "source": [ + "best = (None, None, None, 0.0)\n", + "for trial in job_response.trials:\n", + " print(trial)\n", + " # Keep track of the best outcome\n", + " if float(trial.final_measurement.metrics[0].value) > best[3]:\n", + " try:\n", + " best = (\n", + " trial.id,\n", + " float(trial.parameters[0].value),\n", + " float(trial.parameters[1].value),\n", + " float(trial.final_measurement.metrics[0].value),\n", + " )\n", + " except:\n", + " best = (\n", + " trial.id,\n", + " float(trial.parameters[0].value),\n", + " None,\n", + " float(trial.final_measurement.metrics[0].value),\n", + " )" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "best_trial" + }, + "source": [ + "### Best trial\n", + "\n", + "Now look at which trial was the best:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "best_trial" + }, + "outputs": [], + "source": [ + "print(\"ID\", best[0])\n", + "print(\"Learning Rate\", best[1])\n", + "print(\"Decay\", best[2])\n", + "print(\"Validation Accuracy\", best[3])" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_best_model" + }, + "source": [ + "## Get the Best Model\n", + "\n", + "If you used the method of having the service tell the tuning script where to save the model artifacts (`DIRECT = False`), then the model artifacts for the best model are saved at:\n", + "\n", + " MODEL_DIR//model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_best_model" + }, + "outputs": [], + "source": [ + "BEST_MODEL_DIR = MODEL_DIR + \"/\" + best[0] + \"/model\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tune_custom_model:random" + }, + "source": [ + "## Tuning a model - CIFAR10\n", + "\n", + "Now that you have seen the overall steps for hyperparameter tuning a custom training job using a Python package that mimics training a model, you will do a new hyperparameter tuning job for a custom training job for a CIFAR10 model.\n", + "\n", + "For this example, you will change two parts:\n", + "\n", + "1. Specify the CIFAR10 custom hyperparameter tuning Python package.\n", + "2. Specify a study specification specific to the hyperparameters used in the CIFAR10 custom hyperparameter tuning Python package." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_study_spec:random" + }, + "source": [ + "### Create a study specification\n", + "\n", + "In this study, you will tune for two hyperparameters using the random search algorithm:\n", + "\n", + "- **learning rate**: The search space is a set of discrete values.\n", + "- **learning rate decay**: The search space is a continuous range between 1e-6 and 1e-2.\n", + "\n", + "The objective (goal) is to maximize the validation accuracy.\n", + "\n", + "You will run a maximum of six trials." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_study_spec:random" + }, + "outputs": [], + "source": [ + "study_spec = {\n", + " \"metrics\": [\n", + " {\n", + " \"metric_id\": \"val_accuracy\",\n", + " \"goal\": aip.StudySpec.MetricSpec.GoalType.MAXIMIZE,\n", + " }\n", + " ],\n", + " \"parameters\": [\n", + " {\n", + " \"parameter_id\": \"lr\",\n", + " \"discrete_value_spec\": {\"values\": [0.001, 0.01, 0.1]},\n", + " \"scale_type\": aip.StudySpec.ParameterSpec.ScaleType.UNIT_LINEAR_SCALE,\n", + " },\n", + " {\n", + " \"parameter_id\": \"decay\",\n", + " \"double_value_spec\": {\"min_value\": 1e-6, \"max_value\": 1e-2},\n", + " \"scale_type\": aip.StudySpec.ParameterSpec.ScaleType.UNIT_LINEAR_SCALE,\n", + " },\n", + " ],\n", + " \"algorithm\": aip.StudySpec.Algorithm.RANDOM_SEARCH,\n", + "}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "assemble_custom_hpt_job_specification" + }, + "source": [ + "### Assemble a hyperparameter tuning job specification\n", + "\n", + "Now assemble the complete description for the custom hyperparameter tuning specification:\n", + "\n", + "- `display_name`: The human readable name you assign to this custom hyperparameter tuning job.\n", + "- `trial_job_spec`: The specification for the custom hyperparameter tuning job.\n", + "- `study_spec`: The specification for what to tune.\n", + "- `max_trial_count`: The maximum number of tuning trials.\n", + "- `parallel_trial_count`: How many trials to try in parallel; otherwise, they are done sequentially." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "assemble_custom_hpt_job_specification" + }, + "outputs": [], + "source": [ + "hpt_job = {\n", + " \"display_name\": JOB_NAME,\n", + " \"trial_job_spec\": {\"worker_pool_specs\": worker_pool_spec},\n", + " \"study_spec\": study_spec,\n", + " \"max_trial_count\": 6,\n", + " \"parallel_trial_count\": 1,\n", + "}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:hpt,cifar10" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the hyperparameter tuning script task.py. I won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Parse the command line arguments for the hyperparameter settings for the current trial.\n", + " - Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n", + "- Download and preprocess the CIFAR10 dataset.\n", + "- Build a CNN model.\n", + "- The learning rate and decay hyperparameter values are used during the compile of the model.\n", + "- A definition of a callback `HPTCallback` which obtains the validation accuracy at the end of each epoch (`on_epoch_end()`) and reports it to the hyperparameter tuning service using `hpt.report_hyperparameter_tuning_metric()`.\n", + "- Train the model with the `fit()` method and specify a callback which will report the validation accuracy back to the hyperparameter tuning service." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:hpt,cifar10" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Custom Job for CIFAR10\n", + "\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from hypertune import HyperTune\n", + "import argparse\n", + "import os\n", + "import sys\n", + "\n", + "# Command Line arguments\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv('AIP_MODEL_DIR'), type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.001, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--decay', dest='decay',\n", + " default=0.98, type=float,\n", + " help='Decay rate')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=10, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=200, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "args = parser.parse_args()\n", + "\n", + "\n", + "# Scaling CIFAR-10 data from (0, 255] to (0., 1.]\n", + "def scale(image, label):\n", + " image = tf.cast(image, tf.float32)\n", + " image /= 255.0\n", + " return image, label\n", + "\n", + "\n", + "# Download the dataset\n", + "datasets = tfds.load(name='cifar10', as_supervised=True)\n", + "\n", + "# Preparing dataset\n", + "BUFFER_SIZE = 10000\n", + "BATCH_SIZE = 64\n", + "train_dataset = datasets['train'].map(scale).shuffle(BUFFER_SIZE).batch(BATCH_SIZE)\n", + "test_dataset = datasets['test'].map(scale).batch(BATCH_SIZE)\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_cnn_model():\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu', input_shape=(32, 32, 3)),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu'),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Flatten(),\n", + " tf.keras.layers.Dense(10, activation='softmax')\n", + " ])\n", + " model.compile(\n", + " loss=tf.keras.losses.sparse_categorical_crossentropy,\n", + " optimizer=tf.keras.optimizers.SGD(learning_rate=args.lr, decay=args.decay),\n", + " metrics=['accuracy'])\n", + " return model\n", + "\n", + "\n", + "model = build_and_compile_cnn_model()\n", + "\n", + "# Instantiate the HyperTune reporting object\n", + "hpt = HyperTune()\n", + "\n", + "# Reporting callback\n", + "class HPTCallback(tf.keras.callbacks.Callback):\n", + "\n", + " def on_epoch_end(self, epoch, logs=None):\n", + " global hpt\n", + " hpt.report_hyperparameter_tuning_metric(\n", + " hyperparameter_metric_tag='val_accuracy',\n", + " metric_value=logs['val_accuracy'],\n", + " global_step=epoch)\n", + "\n", + "# Train the model\n", + "model.fit(train_dataset, epochs=5, steps_per_epoch=10, validation_data=test_dataset.take(8),\n", + " callbacks=[HPTCallback()])\n", + "model.save(args.model_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tarball_training_script" + }, + "source": [ + "#### Store hyperparameter tuning script on your Cloud Storage bucket\n", + "\n", + "Next, you package the hyperparameter tuning folder into a compressed tar ball, and then store it in your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tarball_training_script" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_cifar10.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "report_hypertune" + }, + "source": [ + "#### Reporting back the result of the trial using hypertune\n", + "\n", + "For each trial, your Python script needs to report back to the hyperparameter tuning service the objective metric for which you specified as the criteria for evaluating the trial.\n", + "\n", + "For this example, you will specify in the study specification that the objective metric will be reported back as `val_accuracy`.\n", + "\n", + "You report back the value of the objective metric using `HyperTune`. This Python module is used to communicate key/value pairs to the hyperparameter tuning service. To setup this reporting in your Python package, you will add code for the following three steps:\n", + "\n", + "1. Import the HyperTune module: `from hypertune import HyperTune()`.\n", + "2. At the end of every epoch, write the current value of the objective function to the log as a key/value pair using `hpt.report_hyperparameter_tuning_metric()`. In this example, the parameters are:\n", + " - `hyperparameter_metric_tag`: The name of the objective metric to report back. The name must be identical to the name specified in the study specification.\n", + " - `metric_value`: The value of the objective metric to report back to the hyperparameter service.\n", + " - `global_step`: The epoch iteration, starting at 0." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tune_custom_job" + }, + "source": [ + "## Hyperparameter Tune the model\n", + "\n", + "Now start the hyperparameter tuning of your custom model on Vertex. Use this helper function `create_hyperparameter_tuning_job`, which takes the following parameter:\n", + "\n", + "-`hpt_job`: The specification for the hyperparameter tuning job.\n", + "\n", + "The helper function calls job client service's `create_hyperparameter_tuning_job` method, with the following parameters:\n", + "\n", + "-`parent`: The Vertex location path to `Dataset`, `Model` and `Endpoint` resources.\n", + "-`hyperparameter_tuning_job`: The specification for the hyperparameter tuning job.\n", + "\n", + "You will display a handful of the fields returned in `response` object, with the two that are of most interest are:\n", + "\n", + "`response.name`: The Vertex fully qualified identifier assigned to this custom hyperparameter tuning job. You save this identifier for using in subsequent steps.\n", + "\n", + "`response.state`: The current state of the custom hyperparameter tuning job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tune_custom_job" + }, + "outputs": [], + "source": [ + "def create_hyperparameter_tuning_job(hpt_job):\n", + " response = clients[\"job\"].create_hyperparameter_tuning_job(\n", + " parent=PARENT, hyperparameter_tuning_job=hpt_job\n", + " )\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = create_hyperparameter_tuning_job(hpt_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "job_id:response" + }, + "source": [ + "Now get the unique identifier for the custom job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom job\n", + "job_id = response.name\n", + "# The short numeric ID for the custom job\n", + "job_short_id = job_id.split(\"/\")[-1]\n", + "\n", + "print(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_hpt_job" + }, + "source": [ + "### Get information on a hyperparameter tuning job\n", + "\n", + "Next, use this helper function `get_hyperparameter_tuning_job`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the hyperparameter tuning job.\n", + "\n", + "The helper function calls the job client service's `get_hyperparameter_tuning_job` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the hyperparameter tuning job.\n", + "\n", + "If you recall, you got the Vertex fully qualified identifier for the hyperparameter tuning job in the `response.name` field when you called the `create_hyperparameter_tuning_job` method, and saved the identifier in the variable `hpt_job_id`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_hpt_job" + }, + "outputs": [], + "source": [ + "def get_hyperparameter_tuning_job(name, silent=False):\n", + " response = clients[\"job\"].get_hyperparameter_tuning_job(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = get_hyperparameter_tuning_job(hpt_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_tuning_complete" + }, + "source": [ + "## Wait for tuning to complete\n", + "\n", + "Hyperparameter tuning the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done tuning, you can calculate the actual time it took to tune the model by subtracting `end_time` from `start_time`.\n", + "\n", + "For your model, we will need to know the location of the saved models for each trial, which the Python script saved in your local Cloud Storage bucket at `MODEL_DIR + '//saved_model.pb'`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_tuning_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " job_response = get_hyperparameter_tuning_job(hpt_job_id, True)\n", + " if job_response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"Study trials have not completed:\", job_response.state)\n", + " if job_response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " if not DIRECT:\n", + " MODEL_DIR = MODEL_DIR + \"/model\"\n", + " print(\"Study trials have completed\")\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "review_study_results" + }, + "source": [ + "### Review the results of the study\n", + "\n", + "Now review the results of trials." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "review_study_results" + }, + "outputs": [], + "source": [ + "best = (None, None, None, 0.0)\n", + "for trial in job_response.trials:\n", + " print(trial)\n", + " # Keep track of the best outcome\n", + " if float(trial.final_measurement.metrics[0].value) > best[3]:\n", + " try:\n", + " best = (\n", + " trial.id,\n", + " float(trial.parameters[0].value),\n", + " float(trial.parameters[1].value),\n", + " float(trial.final_measurement.metrics[0].value),\n", + " )\n", + " except:\n", + " best = (\n", + " trial.id,\n", + " float(trial.parameters[0].value),\n", + " None,\n", + " float(trial.final_measurement.metrics[0].value),\n", + " )" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "best_trial" + }, + "source": [ + "### Best trial\n", + "\n", + "Now look at which trial was the best:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "best_trial" + }, + "outputs": [], + "source": [ + "print(\"ID\", best[0])\n", + "print(\"Learning Rate\", best[1])\n", + "print(\"Decay\", best[2])\n", + "print(\"Validation Accuracy\", best[3])" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_best_model" + }, + "source": [ + "## Get the Best Model\n", + "\n", + "If you used the method of having the service tell the tuning script where to save the model artifacts (`DIRECT = False`), then the model artifacts for the best model are saved at:\n", + "\n", + " MODEL_DIR//model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_best_model" + }, + "outputs": [], + "source": [ + "BEST_MODEL_DIR = MODEL_DIR + \"/\" + best[0] + \"/model\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "load_saved_model" + }, + "source": [ + "## Load the saved model\n", + "\n", + "Your model is stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Now load it from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n", + "\n", + "To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "load_saved_model" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:image" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load the CIFAR10 test (holdout) data from `tf.keras.datasets`, using the method `load_data()`. This will return the dataset as a tuple of two elements. The first element is the training data and the second is the test data. Each element is also a tuple of two elements: the image data, and the corresponding labels.\n", + "\n", + "You don't need the training data, and hence why we loaded it as `(_, _)`.\n", + "\n", + "Before you can run the data through evaluation, you need to preprocess it:\n", + "\n", + "x_test:\n", + "1. Normalize (rescaling) the pixel data by dividing each pixel by 255. This will replace each single byte integer pixel with a 32-bit floating point number between 0 and 1.\n", + "\n", + "y_test:
\n", + "2. The labels are currently scalar (sparse). If you look back at the `compile()` step in the `trainer/task.py` script, you will find that it was compiled for sparse labels. So we don't need to do anything more." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "evaluate_custom_model:image,cifar10" + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "from tensorflow.keras.datasets import cifar10\n", + "\n", + "(_, _), (x_test, y_test) = cifar10.load_data()\n", + "x_test = (x_test / 255.0).astype(np.float32)\n", + "\n", + "print(x_test.shape, y_test.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "perform_evaluation_custom" + }, + "source": [ + "### Perform the model evaluation\n", + "\n", + "Now evaluate how well the model in the custom job did." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "perform_evaluation_custom" + }, + "outputs": [], + "source": [ + "model.evaluate(x_test, y_test)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_hyperparmeter_tuning_image_classification.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_hyperparmeter_tuning_tabular_regression.ipynb b/notebooks/community/gapic/custom/showcase_hyperparmeter_tuning_tabular_regression.ipynb new file mode 100644 index 000000000..09a914e7e --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_hyperparmeter_tuning_tabular_regression.ipynb @@ -0,0 +1,2104 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Hyperparameter tuning tabular regression model\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom,hpt" + }, + "source": [ + "## Overview\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to do hyperparameter tuning for a custom tabular regression model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:custom,boston,lrg" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Boston Housing Prices dataset](https://www.cs.toronto.edu/~delve/data/boston/bostonDetail.html). The version of the dataset you will use in this tutorial is built into TensorFlow. The trained model predicts the median price of a house in units of 1K USD." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,hpt" + }, + "source": [ + "### Objective\n", + "\n", + "In this notebook, you learn how to create a hyperparameter tuning job for a custom tabular regression model from a Python script in a docker container using the Vertex client library. You can alternatively hyperparameter tune models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create an Vertex hyperparameter turning job for training a custom model.\n", + "- Tune the custom model.\n", + "- Evaluate the study results." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:training" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n", + " TRAIN_GPU, TRAIN_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n", + " )\n", + "else:\n", + " TRAIN_GPU, TRAIN_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:training" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for training.\n", + "\n", + "- Set the variable `TF` to the TensorFlow version of the container image. For example, `2-1` would be version 2.1, and `1-15` would be version 1.15. The following list shows some of the pre-built images available:\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-3:latest`\n", + " - TensorFlow 2.4\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-4:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/training/xgboost-cpu.1-1`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/training/scikit-learn-cpu.0-23:latest`\n", + " - Pytorch\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-5:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-6:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-7:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for training](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:training" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + "else:\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "TRAIN_IMAGE = \"gcr.io/cloud-aiplatform/training/{}:latest\".format(TRAIN_VERSION)\n", + "\n", + "print(\"Training:\", TRAIN_IMAGE, TRAIN_GPU, TRAIN_NGPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:training" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for training.\n", + "\n", + "- Set the variable `TRAIN_COMPUTE` to configure the compute resources for the VMs you will use for for training.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: The following is not supported for training:*\n", + "\n", + " - `standard`: 2 vCPUs\n", + " - `highcpu`: 2, 4 and 8 vCPUs\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:training" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_TRAIN_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Train machine type\", TRAIN_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom,hpt" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own hyperparameter tuning and training of a custom tabular regression." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:custom,hpt" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Model Service for `Model` resources.\n", + "- Job Service for hyperparameter tuning." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:custom,hpt" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"job\"] = create_job_client()\n", + "clients[\"model\"] = create_model_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tune_custom_model:simple" + }, + "source": [ + "## Tuning a model - Hello World\n", + "\n", + "There are two ways you can hyperparameter tune and train a custom model using a container image:\n", + "\n", + "- **Use a Google Cloud prebuilt container**. If you use a prebuilt container, you will additionally specify a Python package to install into the container image. This Python package contains your code for hyperparameter tuning and training a custom model.\n", + "\n", + "- **Use your own custom container image**. If you use your own container, the container needs to contain your code for hyperparameter tuning and training a custom model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_specification:prebuilt_container,hpt" + }, + "source": [ + "## Prepare your hyperparameter tuning job specification\n", + "\n", + "Now that your clients are ready, your first step is to create a Job Specification for your hyperparameter tuning job. The job specification will consist of the following:\n", + "\n", + "- `trial_job_spec`: The specification for the custom job.\n", + " - `worker_pool_spec` : The specification of the type of machine(s) you will use for hyperparameter tuning and how many (single or distributed)\n", + " - `python_package_spec` : The specification of the Python package to be installed with the pre-built container.\n", + "\n", + "- `study_spec`: The specification for what to tune.\n", + " - `parameters`: This is the specification of the hyperparameters that you will tune for the custom training job. It will contain a list of the\n", + " - `metrics`: This is the specification on how to evaluate the result of each tuning trial." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "source": [ + "### Prepare your machine specification\n", + "\n", + "Now define the machine specification for your custom hyperparameter tuning job. This tells Vertex what type of machine instance to provision for the hyperparameter tuning.\n", + " - `machine_type`: The type of GCP instance to provision -- e.g., n1-standard-8.\n", + " - `accelerator_type`: The type, if any, of hardware accelerator. In this tutorial if you previously set the variable `TRAIN_GPU != None`, you are using a GPU; otherwise you will use a CPU.\n", + " - `accelerator_count`: The number of accelerators." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "outputs": [], + "source": [ + "if TRAIN_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": TRAIN_COMPUTE,\n", + " \"accelerator_type\": TRAIN_GPU,\n", + " \"accelerator_count\": TRAIN_NGPU,\n", + " }\n", + "else:\n", + " machine_spec = {\"machine_type\": TRAIN_COMPUTE, \"accelerator_count\": 0}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "source": [ + "### Prepare your disk specification\n", + "\n", + "(optional) Now define the disk specification for your custom hyperparameter tuning job. This tells Vertex what type and size of disk to provision in each machine instance for the hyperparameter tuning.\n", + "\n", + " - `boot_disk_type`: Either SSD or Standard. SSD is faster, and Standard is less expensive. Defaults to SSD.\n", + " - `boot_disk_size_gb`: Size of disk in GB." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "outputs": [], + "source": [ + "DISK_TYPE = \"pd-ssd\" # [ pd-ssd, pd-standard]\n", + "DISK_SIZE = 200 # GB\n", + "\n", + "disk_spec = {\"boot_disk_type\": DISK_TYPE, \"boot_disk_size_gb\": DISK_SIZE}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container" + }, + "source": [ + "### Define the worker pool specification\n", + "\n", + "Next, you define the worker pool specification for your custom hyperparameter tuning job. The worker pool specification will consist of the following:\n", + "\n", + "- `replica_count`: The number of instances to provision of this machine type.\n", + "- `machine_spec`: The hardware specification.\n", + "- `disk_spec` : (optional) The disk storage specification.\n", + "\n", + "- `python_package`: The Python training package to install on the VM instance(s) and which Python module to invoke, along with command line arguments for the Python module.\n", + "\n", + "Let's dive deeper now into the python package specification:\n", + "\n", + "-`executor_image_spec`: This is the docker image which is configured for your custom hyperparameter tuning job.\n", + "\n", + "-`package_uris`: This is a list of the locations (URIs) of your python training packages to install on the provisioned instance. The locations need to be in a Cloud Storage bucket. These can be either individual python files or a zip (archive) of an entire package. In the later case, the job service will unzip (unarchive) the contents into the docker image.\n", + "\n", + "-`python_module`: The Python module (script) to invoke for running the custom hyperparameter tuning job. In this example, you will be invoking `trainer.task.py` -- note that it was not neccessary to append the `.py` suffix.\n", + "\n", + "-`args`: The command line arguments to pass to the corresponding Pythom module. In this example, you will be setting:\n", + " - `\"--model-dir=\" + MODEL_DIR` : The Cloud Storage location where to store the model artifacts. There are two ways to tell the hyperparameter tuning script where to save the model artifacts:\n", + " - direct: You pass the Cloud Storage location as a command line argument to your training script (set variable `DIRECT = True`), or\n", + " - indirect: The service passes the Cloud Storage location as the environment variable `AIP_MODEL_DIR` to your training script (set variable `DIRECT = False`). In this case, you tell the service the model artifact location in the job specification.\n", + " - `\"--epochs=\" + EPOCHS`: The number of epochs for training.\n", + " - `\"--steps=\" + STEPS`: The number of steps (batches) per epoch.\n", + " - `\"--distribute=\" + TRAIN_STRATEGY\"` : The hyperparameter tuning distribution strategy to use for single or distributed hyperparameter tuning.\n", + " - `\"single\"`: single device.\n", + " - `\"mirror\"`: all GPU devices on a single compute instance.\n", + " - `\"multi\"`: all GPU devices on all compute instances." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_job_\" + TIMESTAMP\n", + "MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, JOB_NAME)\n", + "\n", + "if not TRAIN_NGPU or TRAIN_NGPU < 2:\n", + " TRAIN_STRATEGY = \"single\"\n", + "else:\n", + " TRAIN_STRATEGY = \"mirror\"\n", + "\n", + "EPOCHS = 20\n", + "STEPS = 100\n", + "\n", + "DIRECT = True\n", + "if DIRECT:\n", + " CMDARGS = [\n", + " \"--model-dir=\" + MODEL_DIR,\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "else:\n", + " CMDARGS = [\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "\n", + "worker_pool_spec = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": machine_spec,\n", + " \"disk_spec\": disk_spec,\n", + " \"python_package_spec\": {\n", + " \"executor_image_uri\": TRAIN_IMAGE,\n", + " \"package_uris\": [BUCKET_NAME + \"/trainer_boston.tar.gz\"],\n", + " \"python_module\": \"trainer.task\",\n", + " \"args\": CMDARGS,\n", + " },\n", + " }\n", + "]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_study_spec:simple" + }, + "source": [ + "### Create a study specification\n", + "\n", + "Let's start with a simple study. You will just use a single parameter -- the *learning rate*. Since its just one parameter, it doesn't make much sense to do a random search. Instead, we will do a grid search over a range of values.\n", + "\n", + "- `metrics`:\n", + " - `metric_id`: In this example, the objective metric to report back is `'val_accuracy'`\n", + " - `goal`: In this example, the hyperparameter tuning service will evaluate trials to maximize the value of the objective metric.\n", + "- `parameters`: The specification for the hyperparameters to tune.\n", + " - `parameter_id`: The name of the hyperparameter that will be passed to the Python package as a command line argument.\n", + " - `scale_type`: The scale type determines the resolution the hyperparameter tuning service uses when searching over the search space.\n", + " - `UNIT_LINEAR_SCALE`: Uses a resolution that is the same everywhere in the search space.\n", + " - `UNIT_LOG_SCALE`: Values close to the bottom of the search space are further away.\n", + " - `UNIT_REVERSE_LOG_SCALE`: Values close to the top of the search space are further away.\n", + " - **search space**: This is where you will specify the search space of values for the hyperparameter to select for tuning.\n", + " - `integer_value_spec`: Specifies an integer range of values between a `min_value` and `max_value`.\n", + " - `double_value_spec`: Specifies a continuous range of values between a `min_value` and `max_value`.\n", + " - `discrete_value_spec`: Specifies a list of values.\n", + "- `algorithm`: The search method for selecting hyperparameter values per trial:\n", + " - `GRID_SEARCH`: Combinatorically search -- which is used in this example.\n", + " - `RANDOM_SEARCH`: Random search.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_study_spec:simple" + }, + "outputs": [], + "source": [ + "study_spec = {\n", + " \"metrics\": [\n", + " {\n", + " \"metric_id\": \"val_accuracy\",\n", + " \"goal\": aip.StudySpec.MetricSpec.GoalType.MAXIMIZE,\n", + " }\n", + " ],\n", + " \"parameters\": [\n", + " {\n", + " \"parameter_id\": \"lr\",\n", + " \"discrete_value_spec\": {\"values\": [0.001, 0.01, 0.1]},\n", + " \"scale_type\": aip.StudySpec.ParameterSpec.ScaleType.UNIT_LINEAR_SCALE,\n", + " }\n", + " ],\n", + " \"algorithm\": aip.StudySpec.Algorithm.GRID_SEARCH,\n", + "}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "assemble_custom_hpt_job_specification" + }, + "source": [ + "### Assemble a hyperparameter tuning job specification\n", + "\n", + "Now assemble the complete description for the custom hyperparameter tuning specification:\n", + "\n", + "- `display_name`: The human readable name you assign to this custom hyperparameter tuning job.\n", + "- `trial_job_spec`: The specification for the custom hyperparameter tuning job.\n", + "- `study_spec`: The specification for what to tune.\n", + "- `max_trial_count`: The maximum number of tuning trials.\n", + "- `parallel_trial_count`: How many trials to try in parallel; otherwise, they are done sequentially." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "assemble_custom_hpt_job_specification" + }, + "outputs": [], + "source": [ + "hpt_job = {\n", + " \"display_name\": JOB_NAME,\n", + " \"trial_job_spec\": {\"worker_pool_specs\": worker_pool_spec},\n", + " \"study_spec\": study_spec,\n", + " \"max_trial_count\": 6,\n", + " \"parallel_trial_count\": 1,\n", + "}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "examine_training_package" + }, + "source": [ + "### Examine the hyperparameter tuning package\n", + "\n", + "#### Package layout\n", + "\n", + "Before you start the hyperparameter tuning, you will look at how a Python package is assembled for a custom hyperparameter tuning job. When unarchived, the package contains the following directory/file layout.\n", + "\n", + "- PKG-INFO\n", + "- README.md\n", + "- setup.cfg\n", + "- setup.py\n", + "- trainer\n", + " - \\_\\_init\\_\\_.py\n", + " - task.py\n", + "\n", + "The files `setup.cfg` and `setup.py` are the instructions for installing the package into the operating environment of the Docker image.\n", + "\n", + "The file `trainer/task.py` is the Python script for executing the custom hyperparameter tuning job. *Note*, when we referred to it in the worker pool specification, we replace the directory slash with a dot (`trainer.task`) and dropped the file suffix (`.py`).\n", + "\n", + "#### Package Assembly\n", + "\n", + "In the following cells, you will assemble the training package." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "examine_training_package" + }, + "outputs": [], + "source": [ + "# Make folder for Python hyperparameter tuning script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\ntag_build =\\n\\ntag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'tensorflow_datasets==1.3.0',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\nName: Boston Housing tabular regression\\n\\nVersion: 0.0.0\\n\\nSummary: Demostration hyperparameter tuning script\\n\\nHome-page: www.google.com\\n\\nAuthor: Google\\n\\nAuthor-email: aferlitsch@google.com\\n\\nLicense: Public\\n\\nDescription: Demo\\n\\nPlatform: Vertex\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:hpt,simple" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the hyperparameter tuning script task.py. I won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Passes the hyperparameter values for a trial as a command line argument (`parser.add_argument('--lr',...)`)\n", + "- Mimics a training loop, where on each loop (epoch) the variable `accuracy` is set to the loop iteration * the learning rate.\n", + "- Reports back the objective metric `accuracy` back to the hyperparameter tuning service using `report_hyperparameter_tuning_metric()`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:hpt,simple" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# HP Tuning hello world example\n", + "\n", + "from __future__ import absolute_import, division, print_function, unicode_literals\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "from hypertune import HyperTune\n", + "import argparse\n", + "import os\n", + "import sys\n", + "import time\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.001, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=10, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=200, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--model-dir',\n", + " dest='model_dir',\n", + " default='/tmp/saved_model',\n", + " type=str,\n", + " help='Model dir.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "print(device_lib.list_local_devices())\n", + "\n", + "# Instantiate the HyperTune reporting object\n", + "hpt = HyperTune()\n", + "\n", + "for epoch in range(1, args.epochs+1):\n", + " # mimic metric result at the end of an epoch\n", + " acc = args.lr * epoch\n", + " # save the metric result to communicate back to the HPT service\n", + " hpt.report_hyperparameter_tuning_metric(\n", + " hyperparameter_metric_tag='val_accuracy',\n", + " metric_value=acc,\n", + " global_step=epoch)\n", + " print('epoch: {}, accuracy: {}'.format(epoch, acc))\n", + " time.sleep(1)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tarball_training_script" + }, + "source": [ + "#### Store hyperparameter tuning script on your Cloud Storage bucket\n", + "\n", + "Next, you package the hyperparameter tuning folder into a compressed tar ball, and then store it in your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tarball_training_script" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_boston.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "report_hypertune" + }, + "source": [ + "#### Reporting back the result of the trial using hypertune\n", + "\n", + "For each trial, your Python script needs to report back to the hyperparameter tuning service the objective metric for which you specified as the criteria for evaluating the trial.\n", + "\n", + "For this example, you will specify in the study specification that the objective metric will be reported back as `val_loss`.\n", + "\n", + "You report back the value of the objective metric using `HyperTune`. This Python module is used to communicate key/value pairs to the hyperparameter tuning service. To setup this reporting in your Python package, you will add code for the following three steps:\n", + "\n", + "1. Import the HyperTune module: `from hypertune import HyperTune()`.\n", + "2. At the end of every epoch, write the current value of the objective function to the log as a key/value pair using `hpt.report_hyperparameter_tuning_metric()`. In this example, the parameters are:\n", + " - `hyperparameter_metric_tag`: The name of the objective metric to report back. The name must be identical to the name specified in the study specification.\n", + " - `metric_value`: The value of the objective metric to report back to the hyperparameter service.\n", + " - `global_step`: The epoch iteration, starting at 0." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tune_custom_job" + }, + "source": [ + "## Hyperparameter Tune the model\n", + "\n", + "Now start the hyperparameter tuning of your custom model on Vertex. Use this helper function `create_hyperparameter_tuning_job`, which takes the following parameter:\n", + "\n", + "-`hpt_job`: The specification for the hyperparameter tuning job.\n", + "\n", + "The helper function calls job client service's `create_hyperparameter_tuning_job` method, with the following parameters:\n", + "\n", + "-`parent`: The Vertex location path to `Dataset`, `Model` and `Endpoint` resources.\n", + "-`hyperparameter_tuning_job`: The specification for the hyperparameter tuning job.\n", + "\n", + "You will display a handful of the fields returned in `response` object, with the two that are of most interest are:\n", + "\n", + "`response.name`: The Vertex fully qualified identifier assigned to this custom hyperparameter tuning job. You save this identifier for using in subsequent steps.\n", + "\n", + "`response.state`: The current state of the custom hyperparameter tuning job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tune_custom_job" + }, + "outputs": [], + "source": [ + "def create_hyperparameter_tuning_job(hpt_job):\n", + " response = clients[\"job\"].create_hyperparameter_tuning_job(\n", + " parent=PARENT, hyperparameter_tuning_job=hpt_job\n", + " )\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = create_hyperparameter_tuning_job(hpt_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "hpt_job_id:response" + }, + "source": [ + "Now get the unique identifier for the hyperparameter tuning job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "hpt_job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the hyperparameter tuning job\n", + "hpt_job_id = response.name\n", + "# The short numeric ID for the hyperparameter tuning job\n", + "hpt_job_short_id = hpt_job_id.split(\"/\")[-1]\n", + "\n", + "print(hpt_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_hpt_job" + }, + "source": [ + "### Get information on a hyperparameter tuning job\n", + "\n", + "Next, use this helper function `get_hyperparameter_tuning_job`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the hyperparameter tuning job.\n", + "\n", + "The helper function calls the job client service's `get_hyperparameter_tuning_job` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the hyperparameter tuning job.\n", + "\n", + "If you recall, you got the Vertex fully qualified identifier for the hyperparameter tuning job in the `response.name` field when you called the `create_hyperparameter_tuning_job` method, and saved the identifier in the variable `hpt_job_id`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_hpt_job" + }, + "outputs": [], + "source": [ + "def get_hyperparameter_tuning_job(name, silent=False):\n", + " response = clients[\"job\"].get_hyperparameter_tuning_job(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = get_hyperparameter_tuning_job(hpt_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_tuning_complete" + }, + "source": [ + "## Wait for tuning to complete\n", + "\n", + "Hyperparameter tuning the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done tuning, you can calculate the actual time it took to tune the model by subtracting `end_time` from `start_time`.\n", + "\n", + "For your model, we will need to know the location of the saved models for each trial, which the Python script saved in your local Cloud Storage bucket at `MODEL_DIR + '//saved_model.pb'`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_tuning_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " job_response = get_hyperparameter_tuning_job(hpt_job_id, True)\n", + " if job_response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"Study trials have not completed:\", job_response.state)\n", + " if job_response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " if not DIRECT:\n", + " MODEL_DIR = MODEL_DIR + \"/model\"\n", + " print(\"Study trials have completed\")\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "review_study_results" + }, + "source": [ + "### Review the results of the study\n", + "\n", + "Now review the results of trials." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "review_study_results" + }, + "outputs": [], + "source": [ + "best = (None, None, None, 0.0)\n", + "for trial in job_response.trials:\n", + " print(trial)\n", + " # Keep track of the best outcome\n", + " if float(trial.final_measurement.metrics[0].value) > best[3]:\n", + " try:\n", + " best = (\n", + " trial.id,\n", + " float(trial.parameters[0].value),\n", + " float(trial.parameters[1].value),\n", + " float(trial.final_measurement.metrics[0].value),\n", + " )\n", + " except:\n", + " best = (\n", + " trial.id,\n", + " float(trial.parameters[0].value),\n", + " None,\n", + " float(trial.final_measurement.metrics[0].value),\n", + " )" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "best_trial" + }, + "source": [ + "### Best trial\n", + "\n", + "Now look at which trial was the best:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "best_trial" + }, + "outputs": [], + "source": [ + "print(\"ID\", best[0])\n", + "print(\"Learning Rate\", best[1])\n", + "print(\"Decay\", best[2])\n", + "print(\"Validation Accuracy\", best[3])" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_best_model" + }, + "source": [ + "## Get the Best Model\n", + "\n", + "If you used the method of having the service tell the tuning script where to save the model artifacts (`DIRECT = False`), then the model artifacts for the best model are saved at:\n", + "\n", + " MODEL_DIR//model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_best_model" + }, + "outputs": [], + "source": [ + "BEST_MODEL_DIR = MODEL_DIR + \"/\" + best[0] + \"/model\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tune_custom_model:random" + }, + "source": [ + "## Tuning a model - Boston Housing\n", + "\n", + "Now that you have seen the overall steps for hyperparameter tuning a custom training job using a Python package that mimics training a model, you will do a new hyperparameter tuning job for a custom training job for a Boston Housing model.\n", + "\n", + "For this example, you will change two parts:\n", + "\n", + "1. Specify the Boston Housing custom hyperparameter tuning Python package.\n", + "2. Specify a study specification specific to the hyperparameters used in the Boston Housing custom hyperparameter tuning Python package." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_study_spec:random" + }, + "source": [ + "### Create a study specification\n", + "\n", + "In this study, you will tune for two hyperparameters using the random search algorithm:\n", + "\n", + "- **learning rate**: The search space is a set of discrete values.\n", + "- **learning rate decay**: The search space is a continuous range between 1e-6 and 1e-2.\n", + "\n", + "The objective (goal) is to maximize the validation accuracy.\n", + "\n", + "You will run a maximum of six trials." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_study_spec:random" + }, + "outputs": [], + "source": [ + "study_spec = {\n", + " \"metrics\": [\n", + " {\"metric_id\": \"val_loss\", \"goal\": aip.StudySpec.MetricSpec.GoalType.MAXIMIZE}\n", + " ],\n", + " \"parameters\": [\n", + " {\n", + " \"parameter_id\": \"lr\",\n", + " \"discrete_value_spec\": {\"values\": [0.001, 0.01, 0.1]},\n", + " \"scale_type\": aip.StudySpec.ParameterSpec.ScaleType.UNIT_LINEAR_SCALE,\n", + " },\n", + " {\n", + " \"parameter_id\": \"decay\",\n", + " \"double_value_spec\": {\"min_value\": 1e-6, \"max_value\": 1e-2},\n", + " \"scale_type\": aip.StudySpec.ParameterSpec.ScaleType.UNIT_LINEAR_SCALE,\n", + " },\n", + " ],\n", + " \"algorithm\": aip.StudySpec.Algorithm.RANDOM_SEARCH,\n", + "}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "assemble_custom_hpt_job_specification" + }, + "source": [ + "### Assemble a hyperparameter tuning job specification\n", + "\n", + "Now assemble the complete description for the custom hyperparameter tuning specification:\n", + "\n", + "- `display_name`: The human readable name you assign to this custom hyperparameter tuning job.\n", + "- `trial_job_spec`: The specification for the custom hyperparameter tuning job.\n", + "- `study_spec`: The specification for what to tune.\n", + "- `max_trial_count`: The maximum number of tuning trials.\n", + "- `parallel_trial_count`: How many trials to try in parallel; otherwise, they are done sequentially." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "assemble_custom_hpt_job_specification" + }, + "outputs": [], + "source": [ + "hpt_job = {\n", + " \"display_name\": JOB_NAME,\n", + " \"trial_job_spec\": {\"worker_pool_specs\": worker_pool_spec},\n", + " \"study_spec\": study_spec,\n", + " \"max_trial_count\": 6,\n", + " \"parallel_trial_count\": 1,\n", + "}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:hpt,boston" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the hyperparameter tuning script task.py. I won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Parse the command line arguments for the hyperparameter settings for the current trial.\n", + " - Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n", + "- Download and preprocess the Boston Housing dataset.\n", + "- Build a DNN model.\n", + "- The number of units per dense layer and learning rate hyperparameter values are used during the build and compile of the model.\n", + "- A definition of a callback `HPTCallback` which obtains the validation loss at the end of each epoch (`on_epoch_end()`) and reports it to the hyperparameter tuning service using `hpt.report_hyperparameter_tuning_metric()`.\n", + "- Train the model with the `fit()` method and specify a callback which will report the validation loss back to the hyperparameter tuning service." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:hpt,boston" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Custom Training for Boston Housing\n", + "\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "from hypertune import HyperTune\n", + "import numpy as np\n", + "import argparse\n", + "import os\n", + "import sys\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv('AIP_MODEL_DIR'), type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.001, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--decay', dest='decay',\n", + " default=0.98, type=float,\n", + " help='Decay rate')\n", + "parser.add_argument('--units', dest='units',\n", + " default=64, type=int,\n", + " help='Number of units.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=20, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=200, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--param-file', dest='param_file',\n", + " default='/tmp/param.txt', type=str,\n", + " help='Output file for parameters')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "\n", + "\n", + "def make_dataset():\n", + "\n", + " # Scaling Boston Housing data features\n", + " def scale(feature):\n", + " max = np.max(feature)\n", + " feature = (feature / max).astype(np.float)\n", + " return feature, max\n", + "\n", + " (x_train, y_train), (x_test, y_test) = tf.keras.datasets.boston_housing.load_data(\n", + " path=\"boston_housing.npz\", test_split=0.2, seed=113\n", + " )\n", + " params = []\n", + " for _ in range(13):\n", + " x_train[_], max = scale(x_train[_])\n", + " x_test[_], _ = scale(x_test[_])\n", + " params.append(max)\n", + "\n", + " # store the normalization (max) value for each feature\n", + " with tf.io.gfile.GFile(args.param_file, 'w') as f:\n", + " f.write(str(params))\n", + " return (x_train, y_train), (x_test, y_test)\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_dnn_model():\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Dense(args.units, activation='relu', input_shape=(13,)),\n", + " tf.keras.layers.Dense(args.units, activation='relu'),\n", + " tf.keras.layers.Dense(1, activation='linear')\n", + " ])\n", + " model.compile(\n", + " loss='mse',\n", + " optimizer=tf.keras.optimizers.RMSprop(learning_rate=args.lr, decay=args.decay))\n", + " return model\n", + "\n", + "\n", + "model = build_and_compile_dnn_model()\n", + "\n", + "# Instantiate the HyperTune reporting object\n", + "hpt = HyperTune()\n", + "\n", + "# Reporting callback\n", + "class HPTCallback(tf.keras.callbacks.Callback):\n", + "\n", + " def on_epoch_end(self, epoch, logs=None):\n", + " global hpt\n", + " hpt.report_hyperparameter_tuning_metric(\n", + " hyperparameter_metric_tag='val_loss',\n", + " metric_value=logs['val_loss'],\n", + " global_step=epoch)\n", + "\n", + "# Train the model\n", + "BATCH_SIZE = 16\n", + "(x_train, y_train), (x_test, y_test) = make_dataset()\n", + "model.fit(x_train, y_train, epochs=args.epochs, batch_size=BATCH_SIZE, validation_split=0.1, callbacks=[HPTCallback()])\n", + "model.save(args.model_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tarball_training_script" + }, + "source": [ + "#### Store hyperparameter tuning script on your Cloud Storage bucket\n", + "\n", + "Next, you package the hyperparameter tuning folder into a compressed tar ball, and then store it in your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tarball_training_script" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_boston.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "report_hypertune" + }, + "source": [ + "#### Reporting back the result of the trial using hypertune\n", + "\n", + "For each trial, your Python script needs to report back to the hyperparameter tuning service the objective metric for which you specified as the criteria for evaluating the trial.\n", + "\n", + "For this example, you will specify in the study specification that the objective metric will be reported back as `val_loss`.\n", + "\n", + "You report back the value of the objective metric using `HyperTune`. This Python module is used to communicate key/value pairs to the hyperparameter tuning service. To setup this reporting in your Python package, you will add code for the following three steps:\n", + "\n", + "1. Import the HyperTune module: `from hypertune import HyperTune()`.\n", + "2. At the end of every epoch, write the current value of the objective function to the log as a key/value pair using `hpt.report_hyperparameter_tuning_metric()`. In this example, the parameters are:\n", + " - `hyperparameter_metric_tag`: The name of the objective metric to report back. The name must be identical to the name specified in the study specification.\n", + " - `metric_value`: The value of the objective metric to report back to the hyperparameter service.\n", + " - `global_step`: The epoch iteration, starting at 0." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tune_custom_job" + }, + "source": [ + "## Hyperparameter Tune the model\n", + "\n", + "Now start the hyperparameter tuning of your custom model on Vertex. Use this helper function `create_hyperparameter_tuning_job`, which takes the following parameter:\n", + "\n", + "-`hpt_job`: The specification for the hyperparameter tuning job.\n", + "\n", + "The helper function calls job client service's `create_hyperparameter_tuning_job` method, with the following parameters:\n", + "\n", + "-`parent`: The Vertex location path to `Dataset`, `Model` and `Endpoint` resources.\n", + "-`hyperparameter_tuning_job`: The specification for the hyperparameter tuning job.\n", + "\n", + "You will display a handful of the fields returned in `response` object, with the two that are of most interest are:\n", + "\n", + "`response.name`: The Vertex fully qualified identifier assigned to this custom hyperparameter tuning job. You save this identifier for using in subsequent steps.\n", + "\n", + "`response.state`: The current state of the custom hyperparameter tuning job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tune_custom_job" + }, + "outputs": [], + "source": [ + "def create_hyperparameter_tuning_job(hpt_job):\n", + " response = clients[\"job\"].create_hyperparameter_tuning_job(\n", + " parent=PARENT, hyperparameter_tuning_job=hpt_job\n", + " )\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = create_hyperparameter_tuning_job(hpt_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "job_id:response" + }, + "source": [ + "Now get the unique identifier for the custom job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom job\n", + "job_id = response.name\n", + "# The short numeric ID for the custom job\n", + "job_short_id = job_id.split(\"/\")[-1]\n", + "\n", + "print(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_hpt_job" + }, + "source": [ + "### Get information on a hyperparameter tuning job\n", + "\n", + "Next, use this helper function `get_hyperparameter_tuning_job`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the hyperparameter tuning job.\n", + "\n", + "The helper function calls the job client service's `get_hyperparameter_tuning_job` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the hyperparameter tuning job.\n", + "\n", + "If you recall, you got the Vertex fully qualified identifier for the hyperparameter tuning job in the `response.name` field when you called the `create_hyperparameter_tuning_job` method, and saved the identifier in the variable `hpt_job_id`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_hpt_job" + }, + "outputs": [], + "source": [ + "def get_hyperparameter_tuning_job(name, silent=False):\n", + " response = clients[\"job\"].get_hyperparameter_tuning_job(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = get_hyperparameter_tuning_job(hpt_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_tuning_complete" + }, + "source": [ + "## Wait for tuning to complete\n", + "\n", + "Hyperparameter tuning the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done tuning, you can calculate the actual time it took to tune the model by subtracting `end_time` from `start_time`.\n", + "\n", + "For your model, we will need to know the location of the saved models for each trial, which the Python script saved in your local Cloud Storage bucket at `MODEL_DIR + '//saved_model.pb'`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_tuning_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " job_response = get_hyperparameter_tuning_job(hpt_job_id, True)\n", + " if job_response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"Study trials have not completed:\", job_response.state)\n", + " if job_response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " if not DIRECT:\n", + " MODEL_DIR = MODEL_DIR + \"/model\"\n", + " print(\"Study trials have completed\")\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "review_study_results" + }, + "source": [ + "### Review the results of the study\n", + "\n", + "Now review the results of trials." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "review_study_results" + }, + "outputs": [], + "source": [ + "best = (None, None, None, 0.0)\n", + "for trial in job_response.trials:\n", + " print(trial)\n", + " # Keep track of the best outcome\n", + " if float(trial.final_measurement.metrics[0].value) > best[3]:\n", + " try:\n", + " best = (\n", + " trial.id,\n", + " float(trial.parameters[0].value),\n", + " float(trial.parameters[1].value),\n", + " float(trial.final_measurement.metrics[0].value),\n", + " )\n", + " except:\n", + " best = (\n", + " trial.id,\n", + " float(trial.parameters[0].value),\n", + " None,\n", + " float(trial.final_measurement.metrics[0].value),\n", + " )" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "best_trial" + }, + "source": [ + "### Best trial\n", + "\n", + "Now look at which trial was the best:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "best_trial" + }, + "outputs": [], + "source": [ + "print(\"ID\", best[0])\n", + "print(\"Learning Rate\", best[1])\n", + "print(\"Decay\", best[2])\n", + "print(\"Validation Accuracy\", best[3])" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_best_model" + }, + "source": [ + "## Get the Best Model\n", + "\n", + "If you used the method of having the service tell the tuning script where to save the model artifacts (`DIRECT = False`), then the model artifacts for the best model are saved at:\n", + "\n", + " MODEL_DIR//model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_best_model" + }, + "outputs": [], + "source": [ + "BEST_MODEL_DIR = MODEL_DIR + \"/\" + best[0] + \"/model\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "load_saved_model" + }, + "source": [ + "## Load the saved model\n", + "\n", + "Your model is stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Now load it from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n", + "\n", + "To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "load_saved_model" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:tabular" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now let's find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load the Boston Housing test (holdout) data from `tf.keras.datasets`, using the method `load_data()`. This will return the dataset as a tuple of two elements. The first element is the training data and the second is the test data. Each element is also a tuple of two elements: the feature data, and the corresponding labels (median value of owner-occupied home).\n", + "\n", + "You don't need the training data, and hence why we loaded it as `(_, _)`.\n", + "\n", + "Before you can run the data through evaluation, you need to preprocess it:\n", + "\n", + "x_test:\n", + "1. Normalize (rescaling) the data in each column by dividing each value by the maximum value of that column. This will replace each single value with a 32-bit floating point number between 0 and 1." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "evaluate_custom_model:tabular,boston" + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "from tensorflow.keras.datasets import boston_housing\n", + "\n", + "(_, _), (x_test, y_test) = boston_housing.load_data(\n", + " path=\"boston_housing.npz\", test_split=0.2, seed=113\n", + ")\n", + "\n", + "\n", + "def scale(feature):\n", + " max = np.max(feature)\n", + " feature = (feature / max).astype(np.float32)\n", + " return feature\n", + "\n", + "\n", + "# Let's save one data item that has not been scaled\n", + "x_test_notscaled = x_test[0:1].copy()\n", + "\n", + "for _ in range(13):\n", + " x_test[_] = scale(x_test[_])\n", + "x_test = x_test.astype(np.float32)\n", + "\n", + "print(x_test.shape, x_test.dtype, y_test.shape)\n", + "print(\"scaled\", x_test[0])\n", + "print(\"unscaled\", x_test_notscaled)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "perform_evaluation_custom" + }, + "source": [ + "### Perform the model evaluation\n", + "\n", + "Now evaluate how well the model in the custom job did." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "perform_evaluation_custom" + }, + "outputs": [], + "source": [ + "model.evaluate(x_test, y_test)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_hyperparmeter_tuning_tabular_regression.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_hyperparmeter_tuning_text_binary_classification.ipynb b/notebooks/community/gapic/custom/showcase_hyperparmeter_tuning_text_binary_classification.ipynb new file mode 100644 index 000000000..3e460d454 --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_hyperparmeter_tuning_text_binary_classification.ipynb @@ -0,0 +1,2079 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Hyperparameter tuning text binary classification model\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom,hpt" + }, + "source": [ + "## Overview\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to do hyperparameter tuning for a custom text binary classification model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:custom,imdb,tbn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [IMDB Movie Reviews](https://www.tensorflow.org/datasets/catalog/imdb_reviews) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts whether a review is positive or negative in sentiment." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,hpt" + }, + "source": [ + "### Objective\n", + "\n", + "In this notebook, you learn how to create a hyperparameter tuning job for a custom text binary classification model from a Python script in a docker container using the Vertex client library. You can alternatively hyperparameter tune models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create an Vertex hyperparameter turning job for training a custom model.\n", + "- Tune the custom model.\n", + "- Evaluate the study results." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:training" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_GPU\"):\n", + " TRAIN_GPU, TRAIN_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_TRAIN_GPU\")),\n", + " )\n", + "else:\n", + " TRAIN_GPU, TRAIN_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:training" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for training.\n", + "\n", + "- Set the variable `TF` to the TensorFlow version of the container image. For example, `2-1` would be version 2.1, and `1-15` would be version 1.15. The following list shows some of the pre-built images available:\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-3:latest`\n", + " - TensorFlow 2.4\n", + " - `gcr.io/cloud-aiplatform/training/tf-cpu.2-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/tf-gpu.2-4:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/training/xgboost-cpu.1-1`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/training/scikit-learn-cpu.0-23:latest`\n", + " - Pytorch\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-4:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-5:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-6:latest`\n", + " - `gcr.io/cloud-aiplatform/training/pytorch-cpu.1-7:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for training](https://cloud.google.com/vertex-ai/docs/training/pre-built-containers)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:training" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + "else:\n", + " if TRAIN_GPU:\n", + " TRAIN_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " TRAIN_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "TRAIN_IMAGE = \"gcr.io/cloud-aiplatform/training/{}:latest\".format(TRAIN_VERSION)\n", + "\n", + "print(\"Training:\", TRAIN_IMAGE, TRAIN_GPU, TRAIN_NGPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:training" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for training.\n", + "\n", + "- Set the variable `TRAIN_COMPUTE` to configure the compute resources for the VMs you will use for for training.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: The following is not supported for training:*\n", + "\n", + " - `standard`: 2 vCPUs\n", + " - `highcpu`: 2, 4 and 8 vCPUs\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:training" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TRAIN_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_TRAIN_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "TRAIN_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Train machine type\", TRAIN_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom,hpt" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own hyperparameter tuning and training of a custom text binary classification." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:custom,hpt" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Model Service for `Model` resources.\n", + "- Job Service for hyperparameter tuning." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:custom,hpt" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"job\"] = create_job_client()\n", + "clients[\"model\"] = create_model_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tune_custom_model:simple" + }, + "source": [ + "## Tuning a model - Hello World\n", + "\n", + "There are two ways you can hyperparameter tune and train a custom model using a container image:\n", + "\n", + "- **Use a Google Cloud prebuilt container**. If you use a prebuilt container, you will additionally specify a Python package to install into the container image. This Python package contains your code for hyperparameter tuning and training a custom model.\n", + "\n", + "- **Use your own custom container image**. If you use your own container, the container needs to contain your code for hyperparameter tuning and training a custom model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_specification:prebuilt_container,hpt" + }, + "source": [ + "## Prepare your hyperparameter tuning job specification\n", + "\n", + "Now that your clients are ready, your first step is to create a Job Specification for your hyperparameter tuning job. The job specification will consist of the following:\n", + "\n", + "- `trial_job_spec`: The specification for the custom job.\n", + " - `worker_pool_spec` : The specification of the type of machine(s) you will use for hyperparameter tuning and how many (single or distributed)\n", + " - `python_package_spec` : The specification of the Python package to be installed with the pre-built container.\n", + "\n", + "- `study_spec`: The specification for what to tune.\n", + " - `parameters`: This is the specification of the hyperparameters that you will tune for the custom training job. It will contain a list of the\n", + " - `metrics`: This is the specification on how to evaluate the result of each tuning trial." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "source": [ + "### Prepare your machine specification\n", + "\n", + "Now define the machine specification for your custom hyperparameter tuning job. This tells Vertex what type of machine instance to provision for the hyperparameter tuning.\n", + " - `machine_type`: The type of GCP instance to provision -- e.g., n1-standard-8.\n", + " - `accelerator_type`: The type, if any, of hardware accelerator. In this tutorial if you previously set the variable `TRAIN_GPU != None`, you are using a GPU; otherwise you will use a CPU.\n", + " - `accelerator_count`: The number of accelerators." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_machine_specification" + }, + "outputs": [], + "source": [ + "if TRAIN_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": TRAIN_COMPUTE,\n", + " \"accelerator_type\": TRAIN_GPU,\n", + " \"accelerator_count\": TRAIN_NGPU,\n", + " }\n", + "else:\n", + " machine_spec = {\"machine_type\": TRAIN_COMPUTE, \"accelerator_count\": 0}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "source": [ + "### Prepare your disk specification\n", + "\n", + "(optional) Now define the disk specification for your custom hyperparameter tuning job. This tells Vertex what type and size of disk to provision in each machine instance for the hyperparameter tuning.\n", + "\n", + " - `boot_disk_type`: Either SSD or Standard. SSD is faster, and Standard is less expensive. Defaults to SSD.\n", + " - `boot_disk_size_gb`: Size of disk in GB." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_disk_specification" + }, + "outputs": [], + "source": [ + "DISK_TYPE = \"pd-ssd\" # [ pd-ssd, pd-standard]\n", + "DISK_SIZE = 200 # GB\n", + "\n", + "disk_spec = {\"boot_disk_type\": DISK_TYPE, \"boot_disk_size_gb\": DISK_SIZE}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container" + }, + "source": [ + "### Define the worker pool specification\n", + "\n", + "Next, you define the worker pool specification for your custom hyperparameter tuning job. The worker pool specification will consist of the following:\n", + "\n", + "- `replica_count`: The number of instances to provision of this machine type.\n", + "- `machine_spec`: The hardware specification.\n", + "- `disk_spec` : (optional) The disk storage specification.\n", + "\n", + "- `python_package`: The Python training package to install on the VM instance(s) and which Python module to invoke, along with command line arguments for the Python module.\n", + "\n", + "Let's dive deeper now into the python package specification:\n", + "\n", + "-`executor_image_spec`: This is the docker image which is configured for your custom hyperparameter tuning job.\n", + "\n", + "-`package_uris`: This is a list of the locations (URIs) of your python training packages to install on the provisioned instance. The locations need to be in a Cloud Storage bucket. These can be either individual python files or a zip (archive) of an entire package. In the later case, the job service will unzip (unarchive) the contents into the docker image.\n", + "\n", + "-`python_module`: The Python module (script) to invoke for running the custom hyperparameter tuning job. In this example, you will be invoking `trainer.task.py` -- note that it was not neccessary to append the `.py` suffix.\n", + "\n", + "-`args`: The command line arguments to pass to the corresponding Pythom module. In this example, you will be setting:\n", + " - `\"--model-dir=\" + MODEL_DIR` : The Cloud Storage location where to store the model artifacts. There are two ways to tell the hyperparameter tuning script where to save the model artifacts:\n", + " - direct: You pass the Cloud Storage location as a command line argument to your training script (set variable `DIRECT = True`), or\n", + " - indirect: The service passes the Cloud Storage location as the environment variable `AIP_MODEL_DIR` to your training script (set variable `DIRECT = False`). In this case, you tell the service the model artifact location in the job specification.\n", + " - `\"--epochs=\" + EPOCHS`: The number of epochs for training.\n", + " - `\"--steps=\" + STEPS`: The number of steps (batches) per epoch.\n", + " - `\"--distribute=\" + TRAIN_STRATEGY\"` : The hyperparameter tuning distribution strategy to use for single or distributed hyperparameter tuning.\n", + " - `\"single\"`: single device.\n", + " - `\"mirror\"`: all GPU devices on a single compute instance.\n", + " - `\"multi\"`: all GPU devices on all compute instances." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "train_custom_job_worker_pool_specification:prebuilt_container" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_job_\" + TIMESTAMP\n", + "MODEL_DIR = \"{}/{}\".format(BUCKET_NAME, JOB_NAME)\n", + "\n", + "if not TRAIN_NGPU or TRAIN_NGPU < 2:\n", + " TRAIN_STRATEGY = \"single\"\n", + "else:\n", + " TRAIN_STRATEGY = \"mirror\"\n", + "\n", + "EPOCHS = 20\n", + "STEPS = 100\n", + "\n", + "DIRECT = True\n", + "if DIRECT:\n", + " CMDARGS = [\n", + " \"--model-dir=\" + MODEL_DIR,\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "else:\n", + " CMDARGS = [\n", + " \"--epochs=\" + str(EPOCHS),\n", + " \"--steps=\" + str(STEPS),\n", + " \"--distribute=\" + TRAIN_STRATEGY,\n", + " ]\n", + "\n", + "worker_pool_spec = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": machine_spec,\n", + " \"disk_spec\": disk_spec,\n", + " \"python_package_spec\": {\n", + " \"executor_image_uri\": TRAIN_IMAGE,\n", + " \"package_uris\": [BUCKET_NAME + \"/trainer_imdb.tar.gz\"],\n", + " \"python_module\": \"trainer.task\",\n", + " \"args\": CMDARGS,\n", + " },\n", + " }\n", + "]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_study_spec:simple" + }, + "source": [ + "### Create a study specification\n", + "\n", + "Let's start with a simple study. You will just use a single parameter -- the *learning rate*. Since its just one parameter, it doesn't make much sense to do a random search. Instead, we will do a grid search over a range of values.\n", + "\n", + "- `metrics`:\n", + " - `metric_id`: In this example, the objective metric to report back is `'val_accuracy'`\n", + " - `goal`: In this example, the hyperparameter tuning service will evaluate trials to maximize the value of the objective metric.\n", + "- `parameters`: The specification for the hyperparameters to tune.\n", + " - `parameter_id`: The name of the hyperparameter that will be passed to the Python package as a command line argument.\n", + " - `scale_type`: The scale type determines the resolution the hyperparameter tuning service uses when searching over the search space.\n", + " - `UNIT_LINEAR_SCALE`: Uses a resolution that is the same everywhere in the search space.\n", + " - `UNIT_LOG_SCALE`: Values close to the bottom of the search space are further away.\n", + " - `UNIT_REVERSE_LOG_SCALE`: Values close to the top of the search space are further away.\n", + " - **search space**: This is where you will specify the search space of values for the hyperparameter to select for tuning.\n", + " - `integer_value_spec`: Specifies an integer range of values between a `min_value` and `max_value`.\n", + " - `double_value_spec`: Specifies a continuous range of values between a `min_value` and `max_value`.\n", + " - `discrete_value_spec`: Specifies a list of values.\n", + "- `algorithm`: The search method for selecting hyperparameter values per trial:\n", + " - `GRID_SEARCH`: Combinatorically search -- which is used in this example.\n", + " - `RANDOM_SEARCH`: Random search.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_study_spec:simple" + }, + "outputs": [], + "source": [ + "study_spec = {\n", + " \"metrics\": [\n", + " {\n", + " \"metric_id\": \"val_accuracy\",\n", + " \"goal\": aip.StudySpec.MetricSpec.GoalType.MAXIMIZE,\n", + " }\n", + " ],\n", + " \"parameters\": [\n", + " {\n", + " \"parameter_id\": \"lr\",\n", + " \"discrete_value_spec\": {\"values\": [0.001, 0.01, 0.1]},\n", + " \"scale_type\": aip.StudySpec.ParameterSpec.ScaleType.UNIT_LINEAR_SCALE,\n", + " }\n", + " ],\n", + " \"algorithm\": aip.StudySpec.Algorithm.GRID_SEARCH,\n", + "}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "assemble_custom_hpt_job_specification" + }, + "source": [ + "### Assemble a hyperparameter tuning job specification\n", + "\n", + "Now assemble the complete description for the custom hyperparameter tuning specification:\n", + "\n", + "- `display_name`: The human readable name you assign to this custom hyperparameter tuning job.\n", + "- `trial_job_spec`: The specification for the custom hyperparameter tuning job.\n", + "- `study_spec`: The specification for what to tune.\n", + "- `max_trial_count`: The maximum number of tuning trials.\n", + "- `parallel_trial_count`: How many trials to try in parallel; otherwise, they are done sequentially." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "assemble_custom_hpt_job_specification" + }, + "outputs": [], + "source": [ + "hpt_job = {\n", + " \"display_name\": JOB_NAME,\n", + " \"trial_job_spec\": {\"worker_pool_specs\": worker_pool_spec},\n", + " \"study_spec\": study_spec,\n", + " \"max_trial_count\": 6,\n", + " \"parallel_trial_count\": 1,\n", + "}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "examine_training_package" + }, + "source": [ + "### Examine the hyperparameter tuning package\n", + "\n", + "#### Package layout\n", + "\n", + "Before you start the hyperparameter tuning, you will look at how a Python package is assembled for a custom hyperparameter tuning job. When unarchived, the package contains the following directory/file layout.\n", + "\n", + "- PKG-INFO\n", + "- README.md\n", + "- setup.cfg\n", + "- setup.py\n", + "- trainer\n", + " - \\_\\_init\\_\\_.py\n", + " - task.py\n", + "\n", + "The files `setup.cfg` and `setup.py` are the instructions for installing the package into the operating environment of the Docker image.\n", + "\n", + "The file `trainer/task.py` is the Python script for executing the custom hyperparameter tuning job. *Note*, when we referred to it in the worker pool specification, we replace the directory slash with a dot (`trainer.task`) and dropped the file suffix (`.py`).\n", + "\n", + "#### Package Assembly\n", + "\n", + "In the following cells, you will assemble the training package." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "examine_training_package" + }, + "outputs": [], + "source": [ + "# Make folder for Python hyperparameter tuning script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\ntag_build =\\n\\ntag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\nsetuptools.setup(\\n\\n install_requires=[\\n\\n 'tensorflow_datasets==1.3.0',\\n\\n ],\\n\\n packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\nName: IMDB Movie Reviews text binary classification\\n\\nVersion: 0.0.0\\n\\nSummary: Demostration hyperparameter tuning script\\n\\nHome-page: www.google.com\\n\\nAuthor: Google\\n\\nAuthor-email: aferlitsch@google.com\\n\\nLicense: Public\\n\\nDescription: Demo\\n\\nPlatform: Vertex\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:hpt,simple" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the hyperparameter tuning script task.py. I won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Passes the hyperparameter values for a trial as a command line argument (`parser.add_argument('--lr',...)`)\n", + "- Mimics a training loop, where on each loop (epoch) the variable `accuracy` is set to the loop iteration * the learning rate.\n", + "- Reports back the objective metric `accuracy` back to the hyperparameter tuning service using `report_hyperparameter_tuning_metric()`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:hpt,simple" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# HP Tuning hello world example\n", + "\n", + "from __future__ import absolute_import, division, print_function, unicode_literals\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "from hypertune import HyperTune\n", + "import argparse\n", + "import os\n", + "import sys\n", + "import time\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.001, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=10, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=200, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--model-dir',\n", + " dest='model_dir',\n", + " default='/tmp/saved_model',\n", + " type=str,\n", + " help='Model dir.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "print(device_lib.list_local_devices())\n", + "\n", + "# Instantiate the HyperTune reporting object\n", + "hpt = HyperTune()\n", + "\n", + "for epoch in range(1, args.epochs+1):\n", + " # mimic metric result at the end of an epoch\n", + " acc = args.lr * epoch\n", + " # save the metric result to communicate back to the HPT service\n", + " hpt.report_hyperparameter_tuning_metric(\n", + " hyperparameter_metric_tag='val_accuracy',\n", + " metric_value=acc,\n", + " global_step=epoch)\n", + " print('epoch: {}, accuracy: {}'.format(epoch, acc))\n", + " time.sleep(1)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tarball_training_script" + }, + "source": [ + "#### Store hyperparameter tuning script on your Cloud Storage bucket\n", + "\n", + "Next, you package the hyperparameter tuning folder into a compressed tar ball, and then store it in your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tarball_training_script" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_imdb.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "report_hypertune" + }, + "source": [ + "#### Reporting back the result of the trial using hypertune\n", + "\n", + "For each trial, your Python script needs to report back to the hyperparameter tuning service the objective metric for which you specified as the criteria for evaluating the trial.\n", + "\n", + "For this example, you will specify in the study specification that the objective metric will be reported back as `loss`.\n", + "\n", + "You report back the value of the objective metric using `HyperTune`. This Python module is used to communicate key/value pairs to the hyperparameter tuning service. To setup this reporting in your Python package, you will add code for the following three steps:\n", + "\n", + "1. Import the HyperTune module: `from hypertune import HyperTune()`.\n", + "2. At the end of every epoch, write the current value of the objective function to the log as a key/value pair using `hpt.report_hyperparameter_tuning_metric()`. In this example, the parameters are:\n", + " - `hyperparameter_metric_tag`: The name of the objective metric to report back. The name must be identical to the name specified in the study specification.\n", + " - `metric_value`: The value of the objective metric to report back to the hyperparameter service.\n", + " - `global_step`: The epoch iteration, starting at 0." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tune_custom_job" + }, + "source": [ + "## Hyperparameter Tune the model\n", + "\n", + "Now start the hyperparameter tuning of your custom model on Vertex. Use this helper function `create_hyperparameter_tuning_job`, which takes the following parameter:\n", + "\n", + "-`hpt_job`: The specification for the hyperparameter tuning job.\n", + "\n", + "The helper function calls job client service's `create_hyperparameter_tuning_job` method, with the following parameters:\n", + "\n", + "-`parent`: The Vertex location path to `Dataset`, `Model` and `Endpoint` resources.\n", + "-`hyperparameter_tuning_job`: The specification for the hyperparameter tuning job.\n", + "\n", + "You will display a handful of the fields returned in `response` object, with the two that are of most interest are:\n", + "\n", + "`response.name`: The Vertex fully qualified identifier assigned to this custom hyperparameter tuning job. You save this identifier for using in subsequent steps.\n", + "\n", + "`response.state`: The current state of the custom hyperparameter tuning job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tune_custom_job" + }, + "outputs": [], + "source": [ + "def create_hyperparameter_tuning_job(hpt_job):\n", + " response = clients[\"job\"].create_hyperparameter_tuning_job(\n", + " parent=PARENT, hyperparameter_tuning_job=hpt_job\n", + " )\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = create_hyperparameter_tuning_job(hpt_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "hpt_job_id:response" + }, + "source": [ + "Now get the unique identifier for the hyperparameter tuning job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "hpt_job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the hyperparameter tuning job\n", + "hpt_job_id = response.name\n", + "# The short numeric ID for the hyperparameter tuning job\n", + "hpt_job_short_id = hpt_job_id.split(\"/\")[-1]\n", + "\n", + "print(hpt_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_hpt_job" + }, + "source": [ + "### Get information on a hyperparameter tuning job\n", + "\n", + "Next, use this helper function `get_hyperparameter_tuning_job`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the hyperparameter tuning job.\n", + "\n", + "The helper function calls the job client service's `get_hyperparameter_tuning_job` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the hyperparameter tuning job.\n", + "\n", + "If you recall, you got the Vertex fully qualified identifier for the hyperparameter tuning job in the `response.name` field when you called the `create_hyperparameter_tuning_job` method, and saved the identifier in the variable `hpt_job_id`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_hpt_job" + }, + "outputs": [], + "source": [ + "def get_hyperparameter_tuning_job(name, silent=False):\n", + " response = clients[\"job\"].get_hyperparameter_tuning_job(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = get_hyperparameter_tuning_job(hpt_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_tuning_complete" + }, + "source": [ + "## Wait for tuning to complete\n", + "\n", + "Hyperparameter tuning the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done tuning, you can calculate the actual time it took to tune the model by subtracting `end_time` from `start_time`.\n", + "\n", + "For your model, we will need to know the location of the saved models for each trial, which the Python script saved in your local Cloud Storage bucket at `MODEL_DIR + '//saved_model.pb'`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_tuning_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " job_response = get_hyperparameter_tuning_job(hpt_job_id, True)\n", + " if job_response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"Study trials have not completed:\", job_response.state)\n", + " if job_response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " if not DIRECT:\n", + " MODEL_DIR = MODEL_DIR + \"/model\"\n", + " print(\"Study trials have completed\")\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "review_study_results" + }, + "source": [ + "### Review the results of the study\n", + "\n", + "Now review the results of trials." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "review_study_results" + }, + "outputs": [], + "source": [ + "best = (None, None, None, 0.0)\n", + "for trial in job_response.trials:\n", + " print(trial)\n", + " # Keep track of the best outcome\n", + " if float(trial.final_measurement.metrics[0].value) > best[3]:\n", + " try:\n", + " best = (\n", + " trial.id,\n", + " float(trial.parameters[0].value),\n", + " float(trial.parameters[1].value),\n", + " float(trial.final_measurement.metrics[0].value),\n", + " )\n", + " except:\n", + " best = (\n", + " trial.id,\n", + " float(trial.parameters[0].value),\n", + " None,\n", + " float(trial.final_measurement.metrics[0].value),\n", + " )" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "best_trial" + }, + "source": [ + "### Best trial\n", + "\n", + "Now look at which trial was the best:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "best_trial" + }, + "outputs": [], + "source": [ + "print(\"ID\", best[0])\n", + "print(\"Learning Rate\", best[1])\n", + "print(\"Decay\", best[2])\n", + "print(\"Validation Accuracy\", best[3])" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_best_model" + }, + "source": [ + "## Get the Best Model\n", + "\n", + "If you used the method of having the service tell the tuning script where to save the model artifacts (`DIRECT = False`), then the model artifacts for the best model are saved at:\n", + "\n", + " MODEL_DIR//model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_best_model" + }, + "outputs": [], + "source": [ + "BEST_MODEL_DIR = MODEL_DIR + \"/\" + best[0] + \"/model\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tune_custom_model:random" + }, + "source": [ + "## Tuning a model - IMDB Movie Reviews\n", + "\n", + "Now that you have seen the overall steps for hyperparameter tuning a custom training job using a Python package that mimics training a model, you will do a new hyperparameter tuning job for a custom training job for a IMDB Movie Reviews model.\n", + "\n", + "For this example, you will change two parts:\n", + "\n", + "1. Specify the IMDB Movie Reviews custom hyperparameter tuning Python package.\n", + "2. Specify a study specification specific to the hyperparameters used in the IMDB Movie Reviews custom hyperparameter tuning Python package." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_study_spec:random" + }, + "source": [ + "### Create a study specification\n", + "\n", + "In this study, you will tune for two hyperparameters using the random search algorithm:\n", + "\n", + "- **learning rate**: The search space is a set of discrete values.\n", + "- **learning rate decay**: The search space is a continuous range between 1e-6 and 1e-2.\n", + "\n", + "The objective (goal) is to maximize the validation accuracy.\n", + "\n", + "You will run a maximum of six trials." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_study_spec:random" + }, + "outputs": [], + "source": [ + "study_spec = {\n", + " \"metrics\": [\n", + " {\"metric_id\": \"loss\", \"goal\": aip.StudySpec.MetricSpec.GoalType.MAXIMIZE}\n", + " ],\n", + " \"parameters\": [\n", + " {\n", + " \"parameter_id\": \"lr\",\n", + " \"discrete_value_spec\": {\"values\": [0.001, 0.01, 0.1]},\n", + " \"scale_type\": aip.StudySpec.ParameterSpec.ScaleType.UNIT_LINEAR_SCALE,\n", + " },\n", + " {\n", + " \"parameter_id\": \"decay\",\n", + " \"double_value_spec\": {\"min_value\": 1e-6, \"max_value\": 1e-2},\n", + " \"scale_type\": aip.StudySpec.ParameterSpec.ScaleType.UNIT_LINEAR_SCALE,\n", + " },\n", + " ],\n", + " \"algorithm\": aip.StudySpec.Algorithm.RANDOM_SEARCH,\n", + "}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "assemble_custom_hpt_job_specification" + }, + "source": [ + "### Assemble a hyperparameter tuning job specification\n", + "\n", + "Now assemble the complete description for the custom hyperparameter tuning specification:\n", + "\n", + "- `display_name`: The human readable name you assign to this custom hyperparameter tuning job.\n", + "- `trial_job_spec`: The specification for the custom hyperparameter tuning job.\n", + "- `study_spec`: The specification for what to tune.\n", + "- `max_trial_count`: The maximum number of tuning trials.\n", + "- `parallel_trial_count`: How many trials to try in parallel; otherwise, they are done sequentially." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "assemble_custom_hpt_job_specification" + }, + "outputs": [], + "source": [ + "hpt_job = {\n", + " \"display_name\": JOB_NAME,\n", + " \"trial_job_spec\": {\"worker_pool_specs\": worker_pool_spec},\n", + " \"study_spec\": study_spec,\n", + " \"max_trial_count\": 6,\n", + " \"parallel_trial_count\": 1,\n", + "}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:hpt,imdb" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the hyperparameter tuning script task.py. I won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Parse the command line arguments for the hyperparameter settings for the current trial.\n", + " - Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n", + "- Loads IMDB Movie Review dataset from TF Datasets (tfds).\n", + "- Builds a simple RNN model using TF.Keras model API.\n", + "- The learning rate and number of units per dense and LSTM layer hyperparameter values are used during the compile of the model.\n", + "- Compiles the model (`compile()`).\n", + "- A definition of a callback `HPTCallback` which obtains the validation loss at the end of each epoch (`on_epoch_end()`) and reports it to the hyperparameter tuning service using `hpt.report_hyperparameter_tuning_metric()`.\n", + "- Train the model with the `fit()` method and specify a callback which will report the validation loss back to the hyperparameter tuning service.\n", + "- Saves the trained model (`save(args.model_dir)`) to the specified model directory." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:hpt,imdb" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Custom Training for IMDB Movie Reviews\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "from hypertune import HyperTune\n", + "import argparse\n", + "import os\n", + "import sys\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv('AIP_MODEL_DIR'), type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=1e-4, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--decay', dest='decay',\n", + " default=0.98, type=float,\n", + " help='Decay rate')\n", + "parser.add_argument('--units', dest='units',\n", + " default=64, type=int,\n", + " help='Number of units.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=20, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=200, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "print(device_lib.list_local_devices())\n", + "\n", + "# Preparing dataset\n", + "BUFFER_SIZE = 1000\n", + "BATCH_SIZE = 64\n", + "\n", + "\n", + "def make_datasets():\n", + " dataset, info = tfds.load('imdb_reviews/subwords8k', with_info=True,\n", + " as_supervised=True)\n", + " train_dataset, test_dataset = dataset['train'], dataset['test']\n", + " encoder = info.features['text'].encoder\n", + "\n", + " padded_shapes = ([None],())\n", + " return train_dataset.shuffle(BUFFER_SIZE).padded_batch(BATCH_SIZE, padded_shapes), encoder\n", + "\n", + "\n", + "train_dataset, encoder = make_datasets()\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_rnn_model(encoder):\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Embedding(encoder.vocab_size, 64),\n", + " tf.keras.layers.Bidirectional(tf.keras.layers.LSTM(args.units)),\n", + " tf.keras.layers.Dense(args.units, activation='relu'),\n", + " tf.keras.layers.Dense(1)\n", + " ])\n", + " model.compile(loss=tf.keras.losses.BinaryCrossentropy(from_logits=True),\n", + " optimizer=tf.keras.optimizers.Adam(learning_rate=args.lr, decay=args.decay),\n", + " metrics=['accuracy'])\n", + " return model\n", + "\n", + "\n", + "model = build_and_compile_rnn_model(encoder)\n", + "\n", + "# Instantiate the HyperTune reporting object\n", + "hpt = HyperTune()\n", + "\n", + "\n", + "# Reporting callback\n", + "class HPTCallback(tf.keras.callbacks.Callback):\n", + "\n", + " def on_epoch_end(self, epoch, logs=None):\n", + " global hpt\n", + " hpt.report_hyperparameter_tuning_metric(\n", + " hyperparameter_metric_tag='loss',\n", + " metric_value=logs['loss'],\n", + " global_step=epoch)\n", + "\n", + "\n", + "model.fit(train_dataset, epochs=args.epochs, callbacks=[HPTCallback()])\n", + "model.save(args.model_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tarball_training_script" + }, + "source": [ + "#### Store hyperparameter tuning script on your Cloud Storage bucket\n", + "\n", + "Next, you package the hyperparameter tuning folder into a compressed tar ball, and then store it in your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tarball_training_script" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz $BUCKET_NAME/trainer_imdb.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "report_hypertune" + }, + "source": [ + "#### Reporting back the result of the trial using hypertune\n", + "\n", + "For each trial, your Python script needs to report back to the hyperparameter tuning service the objective metric for which you specified as the criteria for evaluating the trial.\n", + "\n", + "For this example, you will specify in the study specification that the objective metric will be reported back as `loss`.\n", + "\n", + "You report back the value of the objective metric using `HyperTune`. This Python module is used to communicate key/value pairs to the hyperparameter tuning service. To setup this reporting in your Python package, you will add code for the following three steps:\n", + "\n", + "1. Import the HyperTune module: `from hypertune import HyperTune()`.\n", + "2. At the end of every epoch, write the current value of the objective function to the log as a key/value pair using `hpt.report_hyperparameter_tuning_metric()`. In this example, the parameters are:\n", + " - `hyperparameter_metric_tag`: The name of the objective metric to report back. The name must be identical to the name specified in the study specification.\n", + " - `metric_value`: The value of the objective metric to report back to the hyperparameter service.\n", + " - `global_step`: The epoch iteration, starting at 0." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tune_custom_job" + }, + "source": [ + "## Hyperparameter Tune the model\n", + "\n", + "Now start the hyperparameter tuning of your custom model on Vertex. Use this helper function `create_hyperparameter_tuning_job`, which takes the following parameter:\n", + "\n", + "-`hpt_job`: The specification for the hyperparameter tuning job.\n", + "\n", + "The helper function calls job client service's `create_hyperparameter_tuning_job` method, with the following parameters:\n", + "\n", + "-`parent`: The Vertex location path to `Dataset`, `Model` and `Endpoint` resources.\n", + "-`hyperparameter_tuning_job`: The specification for the hyperparameter tuning job.\n", + "\n", + "You will display a handful of the fields returned in `response` object, with the two that are of most interest are:\n", + "\n", + "`response.name`: The Vertex fully qualified identifier assigned to this custom hyperparameter tuning job. You save this identifier for using in subsequent steps.\n", + "\n", + "`response.state`: The current state of the custom hyperparameter tuning job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tune_custom_job" + }, + "outputs": [], + "source": [ + "def create_hyperparameter_tuning_job(hpt_job):\n", + " response = clients[\"job\"].create_hyperparameter_tuning_job(\n", + " parent=PARENT, hyperparameter_tuning_job=hpt_job\n", + " )\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = create_hyperparameter_tuning_job(hpt_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "job_id:response" + }, + "source": [ + "Now get the unique identifier for the custom job you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "job_id:response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom job\n", + "job_id = response.name\n", + "# The short numeric ID for the custom job\n", + "job_short_id = job_id.split(\"/\")[-1]\n", + "\n", + "print(job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_hpt_job" + }, + "source": [ + "### Get information on a hyperparameter tuning job\n", + "\n", + "Next, use this helper function `get_hyperparameter_tuning_job`, which takes the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the hyperparameter tuning job.\n", + "\n", + "The helper function calls the job client service's `get_hyperparameter_tuning_job` method, with the following parameter:\n", + "\n", + "- `name`: The Vertex fully qualified identifier for the hyperparameter tuning job.\n", + "\n", + "If you recall, you got the Vertex fully qualified identifier for the hyperparameter tuning job in the `response.name` field when you called the `create_hyperparameter_tuning_job` method, and saved the identifier in the variable `hpt_job_id`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_hpt_job" + }, + "outputs": [], + "source": [ + "def get_hyperparameter_tuning_job(name, silent=False):\n", + " response = clients[\"job\"].get_hyperparameter_tuning_job(name=name)\n", + " if silent:\n", + " return response\n", + "\n", + " print(\"name:\", response.name)\n", + " print(\"display_name:\", response.display_name)\n", + " print(\"state:\", response.state)\n", + " print(\"create_time:\", response.create_time)\n", + " print(\"update_time:\", response.update_time)\n", + " return response\n", + "\n", + "\n", + "response = get_hyperparameter_tuning_job(hpt_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_tuning_complete" + }, + "source": [ + "## Wait for tuning to complete\n", + "\n", + "Hyperparameter tuning the above model may take upwards of 20 minutes time.\n", + "\n", + "Once your model is done tuning, you can calculate the actual time it took to tune the model by subtracting `end_time` from `start_time`.\n", + "\n", + "For your model, we will need to know the location of the saved models for each trial, which the Python script saved in your local Cloud Storage bucket at `MODEL_DIR + '//saved_model.pb'`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_tuning_complete" + }, + "outputs": [], + "source": [ + "while True:\n", + " job_response = get_hyperparameter_tuning_job(hpt_job_id, True)\n", + " if job_response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"Study trials have not completed:\", job_response.state)\n", + " if job_response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " if not DIRECT:\n", + " MODEL_DIR = MODEL_DIR + \"/model\"\n", + " print(\"Study trials have completed\")\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "review_study_results" + }, + "source": [ + "### Review the results of the study\n", + "\n", + "Now review the results of trials." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "review_study_results" + }, + "outputs": [], + "source": [ + "best = (None, None, None, 0.0)\n", + "for trial in job_response.trials:\n", + " print(trial)\n", + " # Keep track of the best outcome\n", + " if float(trial.final_measurement.metrics[0].value) > best[3]:\n", + " try:\n", + " best = (\n", + " trial.id,\n", + " float(trial.parameters[0].value),\n", + " float(trial.parameters[1].value),\n", + " float(trial.final_measurement.metrics[0].value),\n", + " )\n", + " except:\n", + " best = (\n", + " trial.id,\n", + " float(trial.parameters[0].value),\n", + " None,\n", + " float(trial.final_measurement.metrics[0].value),\n", + " )" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "best_trial" + }, + "source": [ + "### Best trial\n", + "\n", + "Now look at which trial was the best:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "best_trial" + }, + "outputs": [], + "source": [ + "print(\"ID\", best[0])\n", + "print(\"Learning Rate\", best[1])\n", + "print(\"Decay\", best[2])\n", + "print(\"Validation Accuracy\", best[3])" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_best_model" + }, + "source": [ + "## Get the Best Model\n", + "\n", + "If you used the method of having the service tell the tuning script where to save the model artifacts (`DIRECT = False`), then the model artifacts for the best model are saved at:\n", + "\n", + " MODEL_DIR//model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_best_model" + }, + "outputs": [], + "source": [ + "BEST_MODEL_DIR = MODEL_DIR + \"/\" + best[0] + \"/model\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "load_saved_model" + }, + "source": [ + "## Load the saved model\n", + "\n", + "Your model is stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Now load it from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n", + "\n", + "To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "load_saved_model" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:text" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now let's find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load the IMDB Movie Review test (holdout) data from `tfds.datasets`, using the method `load()`. This will return the dataset as a tuple of two elements. The first element is the dataset and the second is information on the dataset, which will contain the predefined vocabulary encoder. The encoder will convert words into a numerical embedding, which was pretrained and used in the custom training script.\n", + "\n", + "\n", + "When you trained the model, you needed to set a fix input length for your text. For forward feeding batches, the `padded_batch()` property of the corresponding `tf.dataset` was set to pad each input sequence into the same shape for a batch.\n", + "\n", + "For the test data, you also need to set the `padded_batch()` property accordingly." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "evaluate_custom_model:text,imdb" + }, + "outputs": [], + "source": [ + "import tensorflow_datasets as tfds\n", + "\n", + "dataset, info = tfds.load(\"imdb_reviews/subwords8k\", with_info=True, as_supervised=True)\n", + "test_dataset = dataset[\"test\"]\n", + "encoder = info.features[\"text\"].encoder\n", + "\n", + "BATCH_SIZE = 64\n", + "padded_shapes = ([None], ())\n", + "test_dataset = test_dataset.padded_batch(BATCH_SIZE, padded_shapes)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "perform_evaluation_custom" + }, + "source": [ + "### Perform the model evaluation\n", + "\n", + "Now evaluate how well the model in the custom job did." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "perform_evaluation_custom" + }, + "outputs": [], + "source": [ + "model.evaluate(x_test, y_test)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_hyperparmeter_tuning_text_binary_classification.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_local_image_classification_online.ipynb b/notebooks/community/gapic/custom/showcase_local_image_classification_online.ipynb new file mode 100644 index 000000000..0b50f3da1 --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_local_image_classification_online.ipynb @@ -0,0 +1,1737 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Local image classification model for online prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom,local" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to deploy a locally trained custom image classification model for online prediction." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:custom,cifar10,icn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [CIFAR10 dataset](https://www.tensorflow.org/datasets/catalog/cifar10) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use is built into TensorFlow. The trained model predicts which type of class an image is from ten classes: airplane, automobile, bird, cat, deer, dog, frog, horse, ship, truck." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,local" + }, + "source": [ + "### Objective\n", + "\n", + "In this notebook, you create a custom model locally in the notebook, then learn to deploy the locally trained model to Vertex, and then do a prediction on the deployed model. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a model locally.\n", + "- Train the model locally.\n", + "- View the model evaluation.\n", + "- Upload the model as a Vertex `Model` resource.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:prediction,cpu" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for prediction.\n", + "\n", + "Set the variable `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:prediction,cpu" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (None, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:prediction" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for prediction\n", + "\n", + "- Set the variable `TF` to the TensorFlow version of the container image. For example, `2-1` would be version 2.1, and `1-15` would be version 1.15. The following list shows some of the pre-built images available:\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-3:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-90:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-82:latest`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-23:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-22:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-20:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for prediction](https://cloud.google.com/vertex-ai/docs/predictions/pre-built-containers)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf2-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf2-cpu.{}\".format(TF)\n", + "else:\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "DEPLOY_IMAGE = \"gcr.io/cloud-aiplatform/prediction/{}:latest\".format(DEPLOY_VERSION)\n", + "\n", + "print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for prediction.\n", + "\n", + "- Set the variable `DEPLOY_COMPUTE` to configure the compute resources for the VM you will use for prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom,local" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start locally training a custom model CIFAR10, and then deploy the model to the cloud." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:local" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Model Service for `Model` resources.\n", + "- Endpoint Service for deployment.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:local" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_model:local" + }, + "source": [ + "## Train a model locally\n", + "\n", + "In this tutorial, you train a CIFAR10 model locally." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "set_model_dir:local" + }, + "source": [ + "### Set location to store trained model\n", + "\n", + "You set the variable `MODEL_DIR` for where in your Cloud Storage bucket to save the model in TensorFlow SavedModel format.\n", + "\n", + "Also, you create a local folder for the training script." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_model_dir:local" + }, + "outputs": [], + "source": [ + "MODEL_DIR = BUCKET_NAME + \"/cifar10\"\n", + "model_path_to_deploy = MODEL_DIR\n", + "\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "! mkdir custom/trainer" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:cifar10" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the training script task.py. We won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n", + "- Loads CIFAR10 dataset from TF Datasets (tfds).\n", + "- Builds a model using TF.Keras model API.\n", + "- Compiles the model (`compile()`).\n", + "- Sets a training distribution strategy according to the argument `args.distribute`.\n", + "- Trains the model (`fit()`) with epochs and steps according to the arguments `args.epochs` and `args.steps`\n", + "- Saves the trained model (`save(args.model_dir)`) to the specified model directory." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:cifar10" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Single, Mirror and Multi-Machine Distributed Training for CIFAR-10\n", + "\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "import argparse\n", + "import os\n", + "import sys\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv(\"AIP_MODEL_DIR\"), type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.01, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=10, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=200, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "print('DEVICES', device_lib.list_local_devices())\n", + "\n", + "# Single Machine, single compute device\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "# Single Machine, multiple compute device\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "# Multiple Machine, multiple compute device\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "# Multi-worker configuration\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "# Preparing dataset\n", + "BUFFER_SIZE = 10000\n", + "BATCH_SIZE = 64\n", + "\n", + "\n", + "def make_datasets_unbatched():\n", + "\n", + " # Scaling CIFAR10 data from (0, 255] to (0., 1.]\n", + " def scale(image, label):\n", + " image = tf.cast(image, tf.float32)\n", + " image /= 255.0\n", + " return image, label\n", + "\n", + "\n", + " datasets, info = tfds.load(name='cifar10',\n", + " with_info=True,\n", + " as_supervised=True)\n", + " return datasets['train'].map(scale).cache().shuffle(BUFFER_SIZE).repeat()\n", + "\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_cnn_model():\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu', input_shape=(32, 32, 3)),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu'),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Flatten(),\n", + " tf.keras.layers.Dense(10, activation='softmax')\n", + " ])\n", + " model.compile(\n", + " loss=tf.keras.losses.sparse_categorical_crossentropy,\n", + " optimizer=tf.keras.optimizers.SGD(learning_rate=args.lr),\n", + " metrics=['accuracy'])\n", + " return model\n", + "\n", + "\n", + "# Train the model\n", + "NUM_WORKERS = strategy.num_replicas_in_sync\n", + "# Here the batch size scales up by number of workers since\n", + "# `tf.data.Dataset.batch` expects the global batch size.\n", + "GLOBAL_BATCH_SIZE = BATCH_SIZE * NUM_WORKERS\n", + "train_dataset = make_datasets_unbatched().batch(GLOBAL_BATCH_SIZE)\n", + "\n", + "with strategy.scope():\n", + " # Creation of dataset, and model building/compiling need to be within\n", + " # `strategy.scope()`.\n", + " model = build_and_compile_cnn_model()\n", + "\n", + "model.fit(x=train_dataset, epochs=args.epochs, steps_per_epoch=args.steps)\n", + "model.save(args.model_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "local_train_python" + }, + "source": [ + "### Train the model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "local_train_python" + }, + "outputs": [], + "source": [ + "! python custom/trainer/task.py --epochs=10 --model-dir=$MODEL_DIR" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "load_saved_model" + }, + "source": [ + "## Load the saved model\n", + "\n", + "Your model is stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Now load it from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n", + "\n", + "To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "load_saved_model" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:image" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load the CIFAR10 test (holdout) data from `tf.keras.datasets`, using the method `load_data()`. This will return the dataset as a tuple of two elements. The first element is the training data and the second is the test data. Each element is also a tuple of two elements: the image data, and the corresponding labels.\n", + "\n", + "You don't need the training data, and hence why we loaded it as `(_, _)`.\n", + "\n", + "Before you can run the data through evaluation, you need to preprocess it:\n", + "\n", + "x_test:\n", + "1. Normalize (rescaling) the pixel data by dividing each pixel by 255. This will replace each single byte integer pixel with a 32-bit floating point number between 0 and 1.\n", + "\n", + "y_test:
\n", + "2. The labels are currently scalar (sparse). If you look back at the `compile()` step in the `trainer/task.py` script, you will find that it was compiled for sparse labels. So we don't need to do anything more." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "evaluate_custom_model:image,cifar10" + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "from tensorflow.keras.datasets import cifar10\n", + "\n", + "(_, _), (x_test, y_test) = cifar10.load_data()\n", + "x_test = (x_test / 255.0).astype(np.float32)\n", + "\n", + "print(x_test.shape, y_test.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "perform_evaluation_custom" + }, + "source": [ + "### Perform the model evaluation\n", + "\n", + "Now evaluate how well the model in the custom job did." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "perform_evaluation_custom" + }, + "outputs": [], + "source": [ + "model.evaluate(x_test, y_test)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "how_serving_function_works" + }, + "source": [ + "## Upload the model for serving\n", + "\n", + "Next, you will upload your TF.Keras model from the custom job to Vertex `Model` service, which will create a Vertex `Model` resource for your custom model. During upload, you need to define a serving function to convert data to the format your model expects. If you send encoded data to Vertex, your serving function ensures that the data is decoded on the model server before it is passed as input to your model.\n", + "\n", + "### How does the serving function work\n", + "\n", + "When you send a request to an online prediction server, the request is received by a HTTP server. The HTTP server extracts the prediction request from the HTTP request content body. The extracted prediction request is forwarded to the serving function. For Google pre-built prediction containers, the request content is passed to the serving function as a `tf.string`.\n", + "\n", + "The serving function consists of two parts:\n", + "\n", + "- `preprocessing function`:\n", + " - Converts the input (`tf.string`) to the input shape and data type of the underlying model (dynamic graph).\n", + " - Performs the same preprocessing of the data that was done during training the underlying model -- e.g., normalizing, scaling, etc.\n", + "- `post-processing function`:\n", + " - Converts the model output to format expected by the receiving application -- e.q., compresses the output.\n", + " - Packages the output for the the receiving application -- e.g., add headings, make JSON object, etc.\n", + "\n", + "Both the preprocessing and post-processing functions are converted to static graphs which are fused to the model. The output from the underlying model is passed to the post-processing function. The post-processing function passes the converted/packaged output back to the HTTP server. The HTTP server returns the output as the HTTP response content.\n", + "\n", + "One consideration you need to consider when building serving functions for TF.Keras models is that they run as static graphs. That means, you cannot use TF graph operations that require a dynamic graph. If you do, you will get an error during the compile of the serving function which will indicate that you are using an EagerTensor which is not supported." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_image" + }, + "source": [ + "### Serving function for image data\n", + "\n", + "To pass images to the prediction service, you encode the compressed (e.g., JPEG) image bytes into base 64 -- which makes the content safe from modification while transmitting binary data over the network. Since this deployed model expects input data as raw (uncompressed) bytes, you need to ensure that the base 64 encoded data gets converted back to raw bytes before it is passed as input to the deployed model.\n", + "\n", + "To resolve this, define a serving function (`serving_fn`) and attach it to the model as a preprocessing step. Add a `@tf.function` decorator so the serving function is fused to the underlying model (instead of upstream on a CPU).\n", + "\n", + "When you send a prediction or explanation request, the content of the request is base 64 decoded into a Tensorflow string (`tf.string`), which is passed to the serving function (`serving_fn`). The serving function preprocesses the `tf.string` into raw (uncompressed) numpy bytes (`preprocess_fn`) to match the input requirements of the model:\n", + "- `io.decode_jpeg`- Decompresses the JPG image which is returned as a Tensorflow tensor with three channels (RGB).\n", + "- `image.convert_image_dtype` - Changes integer pixel values to float 32.\n", + "- `image.resize` - Resizes the image to match the input shape for the model.\n", + "- `resized / 255.0` - Rescales (normalization) the pixel data between 0 and 1.\n", + "\n", + "At this point, the data can be passed to the model (`m_call`)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_image" + }, + "outputs": [], + "source": [ + "CONCRETE_INPUT = \"numpy_inputs\"\n", + "\n", + "\n", + "def _preprocess(bytes_input):\n", + " decoded = tf.io.decode_jpeg(bytes_input, channels=3)\n", + " decoded = tf.image.convert_image_dtype(decoded, tf.float32)\n", + " resized = tf.image.resize(decoded, size=(32, 32))\n", + " rescale = tf.cast(resized / 255.0, tf.float32)\n", + " return rescale\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def preprocess_fn(bytes_inputs):\n", + " decoded_images = tf.map_fn(\n", + " _preprocess, bytes_inputs, dtype=tf.float32, back_prop=False\n", + " )\n", + " return {\n", + " CONCRETE_INPUT: decoded_images\n", + " } # User needs to make sure the key matches model's input\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def serving_fn(bytes_inputs):\n", + " images = preprocess_fn(bytes_inputs)\n", + " prob = m_call(**images)\n", + " return prob\n", + "\n", + "\n", + "m_call = tf.function(model.call).get_concrete_function(\n", + " [tf.TensorSpec(shape=[None, 32, 32, 3], dtype=tf.float32, name=CONCRETE_INPUT)]\n", + ")\n", + "\n", + "tf.saved_model.save(\n", + " model, model_path_to_deploy, signatures={\"serving_default\": serving_fn}\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_signature:image" + }, + "source": [ + "## Get the serving function signature\n", + "\n", + "You can get the signatures of your model's input and output layers by reloading the model into memory, and querying it for the signatures corresponding to each layer.\n", + "\n", + "For your purpose, you need the signature of the serving function. Why? Well, when we send our data for prediction as a HTTP request packet, the image data is base64 encoded, and our TF.Keras model takes numpy input. Your serving function will do the conversion from base64 to a numpy array.\n", + "\n", + "When making a prediction request, you need to route the request to the serving function instead of the model, so you need to know the input layer name of the serving function -- which you will use later when you make a prediction request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_signature:image" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_path_to_deploy)\n", + "\n", + "serving_input = list(\n", + " loaded.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "print(\"Serving function input:\", serving_input)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "upload_the_model" + }, + "source": [ + "### Upload the model\n", + "\n", + "Use this helper function `upload_model` to upload your model, stored in SavedModel format, up to the `Model` service, which will instantiate a Vertex `Model` resource instance for your model. Once you've done that, you can use the `Model` resource instance in the same way as any other Vertex `Model` resource instance, such as deploying to an `Endpoint` resource for serving predictions.\n", + "\n", + "The helper function takes the following parameters:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` service.\n", + "- `image_uri`: The container image for the model deployment.\n", + "- `model_uri`: The Cloud Storage path to our SavedModel artificat. For this tutorial, this is the Cloud Storage location where the `trainer/task.py` saved the model artifacts, which we specified in the variable `MODEL_DIR`.\n", + "\n", + "The helper function calls the `Model` client service's method `upload_model`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for `Dataset`, `Model` and `Endpoint` resources.\n", + "- `model`: The specification for the Vertex `Model` resource instance.\n", + "\n", + "Let's now dive deeper into the Vertex model specification `model`. This is a dictionary object that consists of the following fields:\n", + "\n", + "- `display_name`: A human readable name for the `Model` resource.\n", + "- `metadata_schema_uri`: Since your model was built without an Vertex `Dataset` resource, you will leave this blank (`''`).\n", + "- `artificat_uri`: The Cloud Storage path where the model is stored in SavedModel format.\n", + "- `container_spec`: This is the specification for the Docker container that will be installed on the `Endpoint` resource, from which the `Model` resource will serve predictions. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + "\n", + "Uploading a model into a Vertex Model resource returns a long running operation, since it may take a few moments. You call response.result(), which is a synchronous call and will return when the Vertex Model resource is ready.\n", + "\n", + "The helper function returns the Vertex fully qualified identifier for the corresponding Vertex Model instance upload_model_response.model. You will save the identifier for subsequent steps in the variable model_to_deploy_id." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "upload_the_model" + }, + "outputs": [], + "source": [ + "IMAGE_URI = DEPLOY_IMAGE\n", + "\n", + "\n", + "def upload_model(display_name, image_uri, model_uri):\n", + " model = {\n", + " \"display_name\": display_name,\n", + " \"metadata_schema_uri\": \"\",\n", + " \"artifact_uri\": model_uri,\n", + " \"container_spec\": {\n", + " \"image_uri\": image_uri,\n", + " \"command\": [],\n", + " \"args\": [],\n", + " \"env\": [{\"name\": \"env_name\", \"value\": \"env_value\"}],\n", + " \"ports\": [{\"container_port\": 8080}],\n", + " \"predict_route\": \"\",\n", + " \"health_route\": \"\",\n", + " },\n", + " }\n", + " response = clients[\"model\"].upload_model(parent=PARENT, model=model)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " upload_model_response = response.result(timeout=180)\n", + " print(\"upload_model_response\")\n", + " print(\" model:\", upload_model_response.model)\n", + " return upload_model_response.model\n", + "\n", + "\n", + "model_to_deploy_id = upload_model(\n", + " \"cifar10-\" + TIMESTAMP, IMAGE_URI, model_path_to_deploy\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_model" + }, + "source": [ + "### Get `Model` resource information\n", + "\n", + "Now let's get the model information for just your model. Use this helper function `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource.\n", + "\n", + "This helper function calls the Vertex `Model` client service's method `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_model" + }, + "outputs": [], + "source": [ + "def get_model(name):\n", + " response = clients[\"model\"].get_model(name=name)\n", + " print(response)\n", + "\n", + "\n", + "get_model(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:custom" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex custom `Model` resource. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"cifar10_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:dedicated" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `dedicated_resources`: This refers to how many compute instances (replicas) that are scaled for serving prediction requests.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `min_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:dedicated" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"cifar10_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " \"machine_spec\": machine_spec,\n", + " },\n", + " \"disable_container_logging\": False,\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_item:test" + }, + "source": [ + "### Get test item\n", + "\n", + "You will use an example out of the test (holdout) portion of the dataset as a test item." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:test" + }, + "outputs": [], + "source": [ + "test_image = x_test[0]\n", + "test_label = y_test[0]\n", + "print(test_image.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "prepare_test_item:test,image" + }, + "source": [ + "### Prepare the request content\n", + "You are going to send the CIFAR10 image as compressed JPG image, instead of the raw uncompressed bytes:\n", + "\n", + "- `cv2.imwrite`: Use openCV to write the uncompressed image to disk as a compressed JPEG image.\n", + " - Denormalize the image data from \\[0,1) range back to [0,255).\n", + " - Convert the 32-bit floating point values to 8-bit unsigned integers.\n", + "- `tf.io.read_file`: Read the compressed JPG images back into memory as raw bytes.\n", + "- `base64.b64encode`: Encode the raw bytes into a base 64 encoded string." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "prepare_test_item:test,image" + }, + "outputs": [], + "source": [ + "import base64\n", + "\n", + "import cv2\n", + "\n", + "cv2.imwrite(\"tmp.jpg\", (test_image * 255).astype(np.uint8))\n", + "\n", + "bytes = tf.io.read_file(\"tmp.jpg\")\n", + "b64str = base64.b64encode(bytes.numpy()).decode(\"utf-8\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "send_prediction_request:image" + }, + "source": [ + "### Send the prediction request\n", + "\n", + "Ok, now you have a test image. Use this helper function `predict_image`, which takes the following parameters:\n", + "\n", + "- `image`: The test image data as a numpy array.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed to.\n", + "- `parameters_dict`: Additional parameters for serving.\n", + "\n", + "This function calls the prediction client service `predict` method with the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed to.\n", + "- `instances`: A list of instances (encoded images) to predict.\n", + "- `parameters`: Additional parameters for serving.\n", + "\n", + "To pass the image data to the prediction service, in the previous step you encoded the bytes into base64 -- which makes the content safe from modification when transmitting binary data over the network. You need to tell the serving binary where your model is deployed to, that the content has been base64 encoded, so it will decode it on the other end in the serving binary.\n", + "\n", + "Each instance in the prediction request is a dictionary entry of the form:\n", + "\n", + " {serving_input: {'b64': content}}\n", + "\n", + "- `input_name`: the name of the input layer of the underlying model.\n", + "- `'b64'`: A key that indicates the content is base64 encoded.\n", + "- `content`: The compressed JPG image bytes as a base64 encoded string.\n", + "\n", + "Since the `predict()` service can take multiple images (instances), you will send your single image as a list of one image. As a final step, you package the instances list into Google's protobuf format -- which is what we pass to the `predict()` service.\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding image in the request. You will see in the output for each prediction:\n", + "\n", + "- `predictions`: Confidence level for the prediction, between 0 and 1, for each of the classes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "send_prediction_request:image" + }, + "outputs": [], + "source": [ + "def predict_image(image, endpoint, parameters_dict):\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [{serving_input: {\"b64\": image}}]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters_dict\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", prediction)\n", + "\n", + "\n", + "predict_image(b64str, endpoint_id, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_local_image_classification_online.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_local_tabular_regression_online.ipynb b/notebooks/community/gapic/custom/showcase_local_tabular_regression_online.ipynb new file mode 100644 index 000000000..18cce3fc8 --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_local_tabular_regression_online.ipynb @@ -0,0 +1,1658 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Local tabular regression model for online prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom,local" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to deploy a locally trained custom tabular regression model for online prediction." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:custom,boston,lrg" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Boston Housing Prices dataset](https://www.cs.toronto.edu/~delve/data/boston/bostonDetail.html). The version of the dataset you will use in this tutorial is built into TensorFlow. The trained model predicts the median price of a house in units of 1K USD." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,local" + }, + "source": [ + "### Objective\n", + "\n", + "In this notebook, you create a custom model locally in the notebook, then learn to deploy the locally trained model to Vertex, and then do a prediction on the deployed model. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a model locally.\n", + "- Train the model locally.\n", + "- View the model evaluation.\n", + "- Upload the model as a Vertex `Model` resource.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:prediction" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for prediction.\n", + "\n", + "Set the variable `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:prediction" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for prediction\n", + "\n", + "- Set the variable `TF` to the TensorFlow version of the container image. For example, `2-1` would be version 2.1, and `1-15` would be version 1.15. The following list shows some of the pre-built images available:\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-3:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-90:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-82:latest`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-23:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-22:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-20:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for prediction](https://cloud.google.com/vertex-ai/docs/predictions/pre-built-containers)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf2-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf2-cpu.{}\".format(TF)\n", + "else:\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "DEPLOY_IMAGE = \"gcr.io/cloud-aiplatform/prediction/{}:latest\".format(DEPLOY_VERSION)\n", + "\n", + "print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for prediction.\n", + "\n", + "- Set the variable `DEPLOY_COMPUTE` to configure the compute resources for the VM you will use for prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom,local" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start locally training a custom model Boston Housing, and then deploy the model to the cloud." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:local" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Model Service for `Model` resources.\n", + "- Endpoint Service for deployment.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:local" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_model:local" + }, + "source": [ + "## Train a model locally\n", + "\n", + "In this tutorial, you train a Boston Housing model locally." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "set_model_dir:local" + }, + "source": [ + "### Set location to store trained model\n", + "\n", + "You set the variable `MODEL_DIR` for where in your Cloud Storage bucket to save the model in TensorFlow SavedModel format.\n", + "\n", + "Also, you create a local folder for the training script." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_model_dir:local" + }, + "outputs": [], + "source": [ + "MODEL_DIR = BUCKET_NAME + \"/boston\"\n", + "model_path_to_deploy = MODEL_DIR\n", + "\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "! mkdir custom/trainer" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:boston" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the training script task.py. I won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Get the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n", + "- Loads Boston Housing dataset from TF.Keras builtin datasets\n", + "- Builds a simple deep neural network model using TF.Keras model API.\n", + "- Compiles the model (`compile()`).\n", + "- Sets a training distribution strategy according to the argument `args.distribute`.\n", + "- Trains the model (`fit()`) with epochs specified by `args.epochs`.\n", + "- Saves the trained model (`save(args.model_dir)`) to the specified model directory.\n", + "- Saves the maximum value for each feature `f.write(str(params))` to the specified parameters file." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:boston" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Single, Mirror and Multi-Machine Distributed Training for Boston Housing\n", + "\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "import numpy as np\n", + "import argparse\n", + "import os\n", + "import sys\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv('AIP_MODEL_DIR'), type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.001, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=20, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=100, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "parser.add_argument('--param-file', dest='param_file',\n", + " default='/tmp/param.txt', type=str,\n", + " help='Output file for parameters')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "\n", + "# Single Machine, single compute device\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "# Single Machine, multiple compute device\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "# Multiple Machine, multiple compute device\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "# Multi-worker configuration\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "\n", + "def make_dataset():\n", + "\n", + " # Scaling Boston Housing data features\n", + " def scale(feature):\n", + " max = np.max(feature)\n", + " feature = (feature / max).astype(np.float)\n", + " return feature, max\n", + "\n", + " (x_train, y_train), (x_test, y_test) = tf.keras.datasets.boston_housing.load_data(\n", + " path=\"boston_housing.npz\", test_split=0.2, seed=113\n", + " )\n", + " params = []\n", + " for _ in range(13):\n", + " x_train[_], max = scale(x_train[_])\n", + " x_test[_], _ = scale(x_test[_])\n", + " params.append(max)\n", + "\n", + " # store the normalization (max) value for each feature\n", + " with tf.io.gfile.GFile(args.param_file, 'w') as f:\n", + " f.write(str(params))\n", + " return (x_train, y_train), (x_test, y_test)\n", + "\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_dnn_model():\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Dense(128, activation='relu', input_shape=(13,)),\n", + " tf.keras.layers.Dense(128, activation='relu'),\n", + " tf.keras.layers.Dense(1, activation='linear')\n", + " ])\n", + " model.compile(\n", + " loss='mse',\n", + " optimizer=tf.keras.optimizers.RMSprop(learning_rate=args.lr))\n", + " return model\n", + "\n", + "NUM_WORKERS = strategy.num_replicas_in_sync\n", + "# Here the batch size scales up by number of workers since\n", + "# `tf.data.Dataset.batch` expects the global batch size.\n", + "BATCH_SIZE = 16\n", + "GLOBAL_BATCH_SIZE = BATCH_SIZE * NUM_WORKERS\n", + "\n", + "with strategy.scope():\n", + " # Creation of dataset, and model building/compiling need to be within\n", + " # `strategy.scope()`.\n", + " model = build_and_compile_dnn_model()\n", + "\n", + "# Train the model\n", + "(x_train, y_train), (x_test, y_test) = make_dataset()\n", + "model.fit(x_train, y_train, epochs=args.epochs, batch_size=GLOBAL_BATCH_SIZE)\n", + "model.save(args.model_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "local_train_python" + }, + "source": [ + "### Train the model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "local_train_python" + }, + "outputs": [], + "source": [ + "! python custom/trainer/task.py --epochs=10 --model-dir=$MODEL_DIR" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "load_saved_model" + }, + "source": [ + "## Load the saved model\n", + "\n", + "Your model is stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Now load it from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n", + "\n", + "To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "load_saved_model" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:tabular" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now let's find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load the Boston Housing test (holdout) data from `tf.keras.datasets`, using the method `load_data()`. This will return the dataset as a tuple of two elements. The first element is the training data and the second is the test data. Each element is also a tuple of two elements: the feature data, and the corresponding labels (median value of owner-occupied home).\n", + "\n", + "You don't need the training data, and hence why we loaded it as `(_, _)`.\n", + "\n", + "Before you can run the data through evaluation, you need to preprocess it:\n", + "\n", + "x_test:\n", + "1. Normalize (rescaling) the data in each column by dividing each value by the maximum value of that column. This will replace each single value with a 32-bit floating point number between 0 and 1." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "evaluate_custom_model:tabular,boston" + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "from tensorflow.keras.datasets import boston_housing\n", + "\n", + "(_, _), (x_test, y_test) = boston_housing.load_data(\n", + " path=\"boston_housing.npz\", test_split=0.2, seed=113\n", + ")\n", + "\n", + "\n", + "def scale(feature):\n", + " max = np.max(feature)\n", + " feature = (feature / max).astype(np.float32)\n", + " return feature\n", + "\n", + "\n", + "# Let's save one data item that has not been scaled\n", + "x_test_notscaled = x_test[0:1].copy()\n", + "\n", + "for _ in range(13):\n", + " x_test[_] = scale(x_test[_])\n", + "x_test = x_test.astype(np.float32)\n", + "\n", + "print(x_test.shape, x_test.dtype, y_test.shape)\n", + "print(\"scaled\", x_test[0])\n", + "print(\"unscaled\", x_test_notscaled)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "perform_evaluation_custom" + }, + "source": [ + "### Perform the model evaluation\n", + "\n", + "Now evaluate how well the model in the custom job did." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "perform_evaluation_custom" + }, + "outputs": [], + "source": [ + "model.evaluate(x_test, y_test)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "how_serving_function_works" + }, + "source": [ + "## Upload the model for serving\n", + "\n", + "Next, you will upload your TF.Keras model from the custom job to Vertex `Model` service, which will create a Vertex `Model` resource for your custom model. During upload, you need to define a serving function to convert data to the format your model expects. If you send encoded data to Vertex, your serving function ensures that the data is decoded on the model server before it is passed as input to your model.\n", + "\n", + "### How does the serving function work\n", + "\n", + "When you send a request to an online prediction server, the request is received by a HTTP server. The HTTP server extracts the prediction request from the HTTP request content body. The extracted prediction request is forwarded to the serving function. For Google pre-built prediction containers, the request content is passed to the serving function as a `tf.string`.\n", + "\n", + "The serving function consists of two parts:\n", + "\n", + "- `preprocessing function`:\n", + " - Converts the input (`tf.string`) to the input shape and data type of the underlying model (dynamic graph).\n", + " - Performs the same preprocessing of the data that was done during training the underlying model -- e.g., normalizing, scaling, etc.\n", + "- `post-processing function`:\n", + " - Converts the model output to format expected by the receiving application -- e.q., compresses the output.\n", + " - Packages the output for the the receiving application -- e.g., add headings, make JSON object, etc.\n", + "\n", + "Both the preprocessing and post-processing functions are converted to static graphs which are fused to the model. The output from the underlying model is passed to the post-processing function. The post-processing function passes the converted/packaged output back to the HTTP server. The HTTP server returns the output as the HTTP response content.\n", + "\n", + "One consideration you need to consider when building serving functions for TF.Keras models is that they run as static graphs. That means, you cannot use TF graph operations that require a dynamic graph. If you do, you will get an error during the compile of the serving function which will indicate that you are using an EagerTensor which is not supported." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_signature" + }, + "source": [ + "## Get the serving function signature\n", + "\n", + "You can get the signatures of your model's input and output layers by reloading the model into memory, and querying it for the signatures corresponding to each layer.\n", + "\n", + "When making a prediction request, you need to route the request to the serving function instead of the model, so you need to know the input layer name of the serving function -- which you will use later when you make a prediction request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_signature" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_path_to_deploy)\n", + "\n", + "serving_input = list(\n", + " loaded.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "print(\"Serving function input:\", serving_input)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "upload_the_model" + }, + "source": [ + "### Upload the model\n", + "\n", + "Use this helper function `upload_model` to upload your model, stored in SavedModel format, up to the `Model` service, which will instantiate a Vertex `Model` resource instance for your model. Once you've done that, you can use the `Model` resource instance in the same way as any other Vertex `Model` resource instance, such as deploying to an `Endpoint` resource for serving predictions.\n", + "\n", + "The helper function takes the following parameters:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` service.\n", + "- `image_uri`: The container image for the model deployment.\n", + "- `model_uri`: The Cloud Storage path to our SavedModel artificat. For this tutorial, this is the Cloud Storage location where the `trainer/task.py` saved the model artifacts, which we specified in the variable `MODEL_DIR`.\n", + "\n", + "The helper function calls the `Model` client service's method `upload_model`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for `Dataset`, `Model` and `Endpoint` resources.\n", + "- `model`: The specification for the Vertex `Model` resource instance.\n", + "\n", + "Let's now dive deeper into the Vertex model specification `model`. This is a dictionary object that consists of the following fields:\n", + "\n", + "- `display_name`: A human readable name for the `Model` resource.\n", + "- `metadata_schema_uri`: Since your model was built without an Vertex `Dataset` resource, you will leave this blank (`''`).\n", + "- `artificat_uri`: The Cloud Storage path where the model is stored in SavedModel format.\n", + "- `container_spec`: This is the specification for the Docker container that will be installed on the `Endpoint` resource, from which the `Model` resource will serve predictions. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + "\n", + "Uploading a model into a Vertex Model resource returns a long running operation, since it may take a few moments. You call response.result(), which is a synchronous call and will return when the Vertex Model resource is ready.\n", + "\n", + "The helper function returns the Vertex fully qualified identifier for the corresponding Vertex Model instance upload_model_response.model. You will save the identifier for subsequent steps in the variable model_to_deploy_id." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "upload_the_model" + }, + "outputs": [], + "source": [ + "IMAGE_URI = DEPLOY_IMAGE\n", + "\n", + "\n", + "def upload_model(display_name, image_uri, model_uri):\n", + " model = {\n", + " \"display_name\": display_name,\n", + " \"metadata_schema_uri\": \"\",\n", + " \"artifact_uri\": model_uri,\n", + " \"container_spec\": {\n", + " \"image_uri\": image_uri,\n", + " \"command\": [],\n", + " \"args\": [],\n", + " \"env\": [{\"name\": \"env_name\", \"value\": \"env_value\"}],\n", + " \"ports\": [{\"container_port\": 8080}],\n", + " \"predict_route\": \"\",\n", + " \"health_route\": \"\",\n", + " },\n", + " }\n", + " response = clients[\"model\"].upload_model(parent=PARENT, model=model)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " upload_model_response = response.result(timeout=180)\n", + " print(\"upload_model_response\")\n", + " print(\" model:\", upload_model_response.model)\n", + " return upload_model_response.model\n", + "\n", + "\n", + "model_to_deploy_id = upload_model(\n", + " \"boston-\" + TIMESTAMP, IMAGE_URI, model_path_to_deploy\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_model" + }, + "source": [ + "### Get `Model` resource information\n", + "\n", + "Now let's get the model information for just your model. Use this helper function `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource.\n", + "\n", + "This helper function calls the Vertex `Model` client service's method `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_model" + }, + "outputs": [], + "source": [ + "def get_model(name):\n", + " response = clients[\"model\"].get_model(name=name)\n", + " print(response)\n", + "\n", + "\n", + "get_model(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:custom" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex custom `Model` resource. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"boston_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:dedicated" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `dedicated_resources`: This refers to how many compute instances (replicas) that are scaled for serving prediction requests.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `min_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:dedicated" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"boston_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " \"machine_spec\": machine_spec,\n", + " },\n", + " \"disable_container_logging\": False,\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_item:test" + }, + "source": [ + "### Get test item\n", + "\n", + "You will use an example out of the test (holdout) portion of the dataset as a test item." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:test,tabular" + }, + "outputs": [], + "source": [ + "test_item = x_test[0]\n", + "test_label = y_test[0]\n", + "print(test_item.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "send_prediction_request:tabular" + }, + "source": [ + "### Send the prediction request\n", + "\n", + "Ok, now you have a test data item. Use this helper function `predict_data`, which takes the parameters:\n", + "\n", + "- `data`: The test data item as a numpy 1D array of floating point values.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `parameters_dict`: Additional parameters for serving.\n", + "\n", + "This function uses the prediction client service and calls the `predict` method with the parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed.\n", + "- `instances`: A list of instances (data items) to predict.\n", + "- `parameters`: Additional parameters for serving.\n", + "\n", + "To pass the test data to the prediction service, you package it for transmission to the serving binary as follows:\n", + "\n", + " 1. Convert the data item from a 1D numpy array to a 1D Python list.\n", + " 2. Convert the prediction request to a serialized Google protobuf (`json_format.ParseDict()`)\n", + "\n", + "\n", + "Each instance in the prediction request is a dictionary entry of the form:\n", + "\n", + " {input_name: content}\n", + "\n", + "- `input_name`: the name of the input layer of the underlying model.\n", + "- `content`: The data item as a 1D Python list.\n", + "\n", + "Since the `predict()` service can take multiple data items (instances), you will send your single data item as a list of one data item. As a final step, you package the instances list into Google's protobuf format -- which is what we pass to the `predict()` service.\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding image in the request. You will see in the output for each prediction:\n", + "\n", + "- `predictions` -- the predicated median value of a house in units of 1K USD." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "send_prediction_request:tabular" + }, + "outputs": [], + "source": [ + "def predict_data(data, endpoint, parameters_dict):\n", + " parameters = json_format.ParseDict(parameters_dict, Value())\n", + "\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [{serving_input: data.tolist()}]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", prediction)\n", + "\n", + "\n", + "predict_data(test_item, endpoint_id, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_local_tabular_regression_online.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_local_text_binary_classification_online.ipynb b/notebooks/community/gapic/custom/showcase_local_text_binary_classification_online.ipynb new file mode 100644 index 000000000..505f7fb2a --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_local_text_binary_classification_online.ipynb @@ -0,0 +1,1635 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: Local text binary classification model for online prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:custom,local" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to deploy a locally trained custom text binary classification model for online prediction." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:custom,imdb,tbn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [IMDB Movie Reviews](https://www.tensorflow.org/datasets/catalog/imdb_reviews) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts whether a review is positive or negative in sentiment." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:custom,local" + }, + "source": [ + "### Objective\n", + "\n", + "In this notebook, you create a custom model locally in the notebook, then learn to deploy the locally trained model to Vertex, and then do a prediction on the deployed model. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a model locally.\n", + "- Train the model locally.\n", + "- View the model evaluation.\n", + "- Upload the model as a Vertex `Model` resource.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:prediction" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for prediction.\n", + "\n", + "Set the variable `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (aip.AcceleratorType.NVIDIA_TESLA_K80, 1)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:prediction" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for prediction\n", + "\n", + "- Set the variable `TF` to the TensorFlow version of the container image. For example, `2-1` would be version 2.1, and `1-15` would be version 1.15. The following list shows some of the pre-built images available:\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-3:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-90:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-82:latest`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-23:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-22:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-20:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for prediction](https://cloud.google.com/vertex-ai/docs/predictions/pre-built-containers)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf2-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf2-cpu.{}\".format(TF)\n", + "else:\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "DEPLOY_IMAGE = \"gcr.io/cloud-aiplatform/prediction/{}:latest\".format(DEPLOY_VERSION)\n", + "\n", + "print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for prediction.\n", + "\n", + "- Set the variable `DEPLOY_COMPUTE` to configure the compute resources for the VM you will use for prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:custom,local" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start locally training a custom model IMDB Movie Reviews, and then deploy the model to the cloud." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:local" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Model Service for `Model` resources.\n", + "- Endpoint Service for deployment.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:local" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_custom_model:local" + }, + "source": [ + "## Train a model locally\n", + "\n", + "In this tutorial, you train a IMDB Movie Reviews model locally." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "set_model_dir:local" + }, + "source": [ + "### Set location to store trained model\n", + "\n", + "You set the variable `MODEL_DIR` for where in your Cloud Storage bucket to save the model in TensorFlow SavedModel format.\n", + "\n", + "Also, you create a local folder for the training script." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_model_dir:local" + }, + "outputs": [], + "source": [ + "MODEL_DIR = BUCKET_NAME + \"/imdb\"\n", + "model_path_to_deploy = MODEL_DIR\n", + "\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "! mkdir custom/trainer" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "taskpy_contents:imdb" + }, + "source": [ + "#### Task.py contents\n", + "\n", + "In the next cell, you write the contents of the training script task.py. I won't go into detail, it's just there for you to browse. In summary:\n", + "\n", + "- Gets the directory where to save the model artifacts from the command line (`--model_dir`), and if not specified, then from the environment variable `AIP_MODEL_DIR`.\n", + "- Loads IMDB Movie Reviews dataset from TF Datasets (tfds).\n", + "- Builds a simple RNN model using TF.Keras model API.\n", + "- Compiles the model (`compile()`).\n", + "- Sets a training distribution strategy according to the argument `args.distribute`.\n", + "- Trains the model (`fit()`) with epochs specified by `args.epochs`.\n", + "- Saves the trained model (`save(args.model_dir)`) to the specified model directory." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "taskpy_contents:imdb" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Single, Mirror and Multi-Machine Distributed Training for IMDB\n", + "\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "import argparse\n", + "import os\n", + "import sys\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv('AIP_MODEL_DIR'), type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=1e-4, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=20, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=100, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "print(device_lib.list_local_devices())\n", + "\n", + "# Single Machine, single compute device\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "# Single Machine, multiple compute device\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "# Multiple Machine, multiple compute device\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "# Multi-worker configuration\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "# Preparing dataset\n", + "BUFFER_SIZE = 10000\n", + "BATCH_SIZE = 64\n", + "\n", + "\n", + "def make_datasets():\n", + " dataset, info = tfds.load('imdb_reviews/subwords8k', with_info=True,\n", + " as_supervised=True)\n", + " train_dataset, test_dataset = dataset['train'], dataset['test']\n", + " encoder = info.features['text'].encoder\n", + "\n", + " padded_shapes = ([None],())\n", + " return train_dataset.shuffle(BUFFER_SIZE).padded_batch(BATCH_SIZE, padded_shapes), encoder\n", + "\n", + "\n", + "train_dataset, encoder = make_datasets()\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_rnn_model(encoder):\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Embedding(encoder.vocab_size, 64),\n", + " tf.keras.layers.Bidirectional(tf.keras.layers.LSTM(64)),\n", + " tf.keras.layers.Dense(64, activation='relu'),\n", + " tf.keras.layers.Dense(1, activation='sigmoid')\n", + " ])\n", + " model.compile(loss=tf.keras.losses.BinaryCrossentropy(from_logits=True),\n", + " optimizer=tf.keras.optimizers.Adam(args.lr),\n", + " metrics=['accuracy'])\n", + " return model\n", + "\n", + "\n", + "with strategy.scope():\n", + " # Creation of dataset, and model building/compiling need to be within\n", + " # `strategy.scope()`.\n", + " model = build_and_compile_rnn_model(encoder)\n", + "\n", + "# Train the model\n", + "model.fit(train_dataset, epochs=args.epochs, steps_per_epoch=args.steps)\n", + "model.save(args.model_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "local_train_python" + }, + "source": [ + "### Train the model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "local_train_python" + }, + "outputs": [], + "source": [ + "! python custom/trainer/task.py --epochs=10 --model-dir=$MODEL_DIR" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "load_saved_model" + }, + "source": [ + "## Load the saved model\n", + "\n", + "Your model is stored in a TensorFlow SavedModel format in a Cloud Storage bucket. Now load it from the Cloud Storage bucket, and then you can do some things, like evaluate the model, and do a prediction.\n", + "\n", + "To load, you use the TF.Keras `model.load_model()` method passing it the Cloud Storage path where the model is saved -- specified by `MODEL_DIR`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "load_saved_model" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(MODEL_DIR)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_custom_model:text" + }, + "source": [ + "## Evaluate the model\n", + "\n", + "Now let's find out how good the model is.\n", + "\n", + "### Load evaluation data\n", + "\n", + "You will load the IMDB Movie Review test (holdout) data from `tfds.datasets`, using the method `load()`. This will return the dataset as a tuple of two elements. The first element is the dataset and the second is information on the dataset, which will contain the predefined vocabulary encoder. The encoder will convert words into a numerical embedding, which was pretrained and used in the custom training script.\n", + "\n", + "\n", + "When you trained the model, you needed to set a fix input length for your text. For forward feeding batches, the `padded_batch()` property of the corresponding `tf.dataset` was set to pad each input sequence into the same shape for a batch.\n", + "\n", + "For the test data, you also need to set the `padded_batch()` property accordingly." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "evaluate_custom_model:text,imdb" + }, + "outputs": [], + "source": [ + "import tensorflow_datasets as tfds\n", + "\n", + "dataset, info = tfds.load(\"imdb_reviews/subwords8k\", with_info=True, as_supervised=True)\n", + "test_dataset = dataset[\"test\"]\n", + "encoder = info.features[\"text\"].encoder\n", + "\n", + "BATCH_SIZE = 64\n", + "padded_shapes = ([None], ())\n", + "test_dataset = test_dataset.padded_batch(BATCH_SIZE, padded_shapes)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "perform_evaluation_custom:tfds" + }, + "source": [ + "### Perform the model evaluation\n", + "\n", + "Now evaluate how well the model in the custom job did." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "perform_evaluation_custom:tfds" + }, + "outputs": [], + "source": [ + "model.evaluate(test_dataset)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "how_serving_function_works" + }, + "source": [ + "## Upload the model for serving\n", + "\n", + "Next, you will upload your TF.Keras model from the custom job to Vertex `Model` service, which will create a Vertex `Model` resource for your custom model. During upload, you need to define a serving function to convert data to the format your model expects. If you send encoded data to Vertex, your serving function ensures that the data is decoded on the model server before it is passed as input to your model.\n", + "\n", + "### How does the serving function work\n", + "\n", + "When you send a request to an online prediction server, the request is received by a HTTP server. The HTTP server extracts the prediction request from the HTTP request content body. The extracted prediction request is forwarded to the serving function. For Google pre-built prediction containers, the request content is passed to the serving function as a `tf.string`.\n", + "\n", + "The serving function consists of two parts:\n", + "\n", + "- `preprocessing function`:\n", + " - Converts the input (`tf.string`) to the input shape and data type of the underlying model (dynamic graph).\n", + " - Performs the same preprocessing of the data that was done during training the underlying model -- e.g., normalizing, scaling, etc.\n", + "- `post-processing function`:\n", + " - Converts the model output to format expected by the receiving application -- e.q., compresses the output.\n", + " - Packages the output for the the receiving application -- e.g., add headings, make JSON object, etc.\n", + "\n", + "Both the preprocessing and post-processing functions are converted to static graphs which are fused to the model. The output from the underlying model is passed to the post-processing function. The post-processing function passes the converted/packaged output back to the HTTP server. The HTTP server returns the output as the HTTP response content.\n", + "\n", + "One consideration you need to consider when building serving functions for TF.Keras models is that they run as static graphs. That means, you cannot use TF graph operations that require a dynamic graph. If you do, you will get an error during the compile of the serving function which will indicate that you are using an EagerTensor which is not supported." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_signature" + }, + "source": [ + "## Get the serving function signature\n", + "\n", + "You can get the signatures of your model's input and output layers by reloading the model into memory, and querying it for the signatures corresponding to each layer.\n", + "\n", + "When making a prediction request, you need to route the request to the serving function instead of the model, so you need to know the input layer name of the serving function -- which you will use later when you make a prediction request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_signature" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_path_to_deploy)\n", + "\n", + "serving_input = list(\n", + " loaded.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "print(\"Serving function input:\", serving_input)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "upload_the_model" + }, + "source": [ + "### Upload the model\n", + "\n", + "Use this helper function `upload_model` to upload your model, stored in SavedModel format, up to the `Model` service, which will instantiate a Vertex `Model` resource instance for your model. Once you've done that, you can use the `Model` resource instance in the same way as any other Vertex `Model` resource instance, such as deploying to an `Endpoint` resource for serving predictions.\n", + "\n", + "The helper function takes the following parameters:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` service.\n", + "- `image_uri`: The container image for the model deployment.\n", + "- `model_uri`: The Cloud Storage path to our SavedModel artificat. For this tutorial, this is the Cloud Storage location where the `trainer/task.py` saved the model artifacts, which we specified in the variable `MODEL_DIR`.\n", + "\n", + "The helper function calls the `Model` client service's method `upload_model`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for `Dataset`, `Model` and `Endpoint` resources.\n", + "- `model`: The specification for the Vertex `Model` resource instance.\n", + "\n", + "Let's now dive deeper into the Vertex model specification `model`. This is a dictionary object that consists of the following fields:\n", + "\n", + "- `display_name`: A human readable name for the `Model` resource.\n", + "- `metadata_schema_uri`: Since your model was built without an Vertex `Dataset` resource, you will leave this blank (`''`).\n", + "- `artificat_uri`: The Cloud Storage path where the model is stored in SavedModel format.\n", + "- `container_spec`: This is the specification for the Docker container that will be installed on the `Endpoint` resource, from which the `Model` resource will serve predictions. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + "\n", + "Uploading a model into a Vertex Model resource returns a long running operation, since it may take a few moments. You call response.result(), which is a synchronous call and will return when the Vertex Model resource is ready.\n", + "\n", + "The helper function returns the Vertex fully qualified identifier for the corresponding Vertex Model instance upload_model_response.model. You will save the identifier for subsequent steps in the variable model_to_deploy_id." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "upload_the_model" + }, + "outputs": [], + "source": [ + "IMAGE_URI = DEPLOY_IMAGE\n", + "\n", + "\n", + "def upload_model(display_name, image_uri, model_uri):\n", + " model = {\n", + " \"display_name\": display_name,\n", + " \"metadata_schema_uri\": \"\",\n", + " \"artifact_uri\": model_uri,\n", + " \"container_spec\": {\n", + " \"image_uri\": image_uri,\n", + " \"command\": [],\n", + " \"args\": [],\n", + " \"env\": [{\"name\": \"env_name\", \"value\": \"env_value\"}],\n", + " \"ports\": [{\"container_port\": 8080}],\n", + " \"predict_route\": \"\",\n", + " \"health_route\": \"\",\n", + " },\n", + " }\n", + " response = clients[\"model\"].upload_model(parent=PARENT, model=model)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " upload_model_response = response.result(timeout=180)\n", + " print(\"upload_model_response\")\n", + " print(\" model:\", upload_model_response.model)\n", + " return upload_model_response.model\n", + "\n", + "\n", + "model_to_deploy_id = upload_model(\"imdb-\" + TIMESTAMP, IMAGE_URI, model_path_to_deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_model" + }, + "source": [ + "### Get `Model` resource information\n", + "\n", + "Now let's get the model information for just your model. Use this helper function `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource.\n", + "\n", + "This helper function calls the Vertex `Model` client service's method `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_model" + }, + "outputs": [], + "source": [ + "def get_model(name):\n", + " response = clients[\"model\"].get_model(name=name)\n", + " print(response)\n", + "\n", + "\n", + "get_model(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:custom" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex custom `Model` resource. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"imdb_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:dedicated" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `dedicated_resources`: This refers to how many compute instances (replicas) that are scaled for serving prediction requests.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `min_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:dedicated" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"imdb_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " \"machine_spec\": machine_spec,\n", + " },\n", + " \"disable_container_logging\": False,\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "prepare_test_item:text" + }, + "source": [ + "### Prepare the request content\n", + "\n", + "Since the dataset is a `tf.dataset`, which acts as a generator, we must use it as an iterator to access the data items in the test data. We do the following to get a single data item from the test data:\n", + "\n", + "- Set the property for the number of batches to draw per iteration to one using the method `take(1)`.\n", + "- Iterate once through the test data -- i.e., we do a break within the for loop.\n", + "- In the single iteration, we save the data item which is in the form of a tuple.\n", + "- The data item will be the first element of the tuple, which you then will convert from an tensor to a numpy array -- `data[0].numpy()`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "prepare_test_item:text" + }, + "outputs": [], + "source": [ + "import tensorflow_datasets as tfds\n", + "\n", + "dataset, info = tfds.load(\"imdb_reviews/subwords8k\", with_info=True, as_supervised=True)\n", + "test_dataset = dataset[\"test\"]\n", + "\n", + "test_dataset.take(1)\n", + "for data in test_dataset:\n", + " print(data)\n", + " break\n", + "\n", + "test_item = data[0].numpy()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "send_prediction_request:text" + }, + "source": [ + "### Send the prediction request\n", + "\n", + "Ok, now you have a test data item. Use this helper function `predict_data`, which takes the following parameters:\n", + "\n", + "- `data`: The test data item is a 64 padded numpy 1D array.\n", + "- `endpoint`: The Vertex AI fully qualified identifier for the endpoint where the model was deployed.\n", + "- `parameters_dict`: Additional parameters for serving.\n", + "\n", + "This function uses the prediction client service and calls the `predict` method with the following parameters:\n", + "\n", + "- `endpoint`: The Vertex AI fully qualified identifier for the endpoint where the model was deployed.\n", + "- `instances`: A list of instances (data items) to predict.\n", + "- `parameters`: Additional parameters for serving.\n", + "\n", + "To pass the test data to the prediction service, you must package it for transmission to the serving binary as follows:\n", + "\n", + " 1. Convert the data item from a 1D numpy array to a 1D Python list.\n", + " 2. Convert the prediction request to a serialized Google protobuf (`json_format.ParseDict()`)\n", + "\n", + "Each instance in the prediction request is a dictionary entry of the form:\n", + "\n", + " {input_name: content}\n", + "\n", + "- `input_name`: the name of the input layer of the underlying model.\n", + "- `content`: The data item as a 1D Python list.\n", + "\n", + "Since the `predict()` service can take multiple data items (instances), you will send your single data item as a list of one data item. As a final step, you package the instances list into Google's protobuf format -- which is what we pass to the `predict()` service.\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding image in the request. You will see in the output for each prediction:\n", + "\n", + "- `predictions` -- the predicated binary sentiment between 0 (negative) and 1 (positive)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "send_prediction_request:text" + }, + "outputs": [], + "source": [ + "def predict_data(data, endpoint, parameters_dict):\n", + " parameters = json_format.ParseDict(parameters_dict, Value())\n", + "\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [{serving_input: data.tolist()}]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", prediction)\n", + "\n", + "\n", + "predict_data(test_item, endpoint_id, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_local_text_binary_classification_online.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/gapic/custom/showcase_tfhub_image_classification_online.ipynb b/notebooks/community/gapic/custom/showcase_tfhub_image_classification_online.ipynb new file mode 100644 index 000000000..e7109a61d --- /dev/null +++ b/notebooks/community/gapic/custom/showcase_tfhub_image_classification_online.ipynb @@ -0,0 +1,1490 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title" + }, + "source": [ + "# Vertex client library: TF Hub image classification model for online prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:tfhub,prediction" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex client library for Python to deploy a pretrained TensorFlow Hub image classification model for online prediction." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:flowers,icn" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Flowers dataset](https://www.tensorflow.org/datasets/catalog/tf_flowers) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts the type of flower an image is from a class of five flowers: daisy, dandelion, rose, sunflower, or tulip." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:tfhub,online_prediction" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you will deploy a TensorFlow Hub pretrained model, and then do a prediction on the deployed model by sending data.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Download a TensorFlow Hub pretrained model.\n", + "- Retrieve and load the model artifacts.\n", + "- Upload the model as a Vertex `Model` resource.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex client library." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = \"--user\"\n", + "else:\n", + " USER_FLAG = \"\"\n", + "\n", + "! pip3 install -U google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex client library and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:custom" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Vertex client library, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex runs\n", + "the code from this package. In this tutorial, Vertex also saves the\n", + "trained model that results from your job in the same bucket. You can then\n", + "create an `Endpoint` resource based on this output in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip:protobuf" + }, + "source": [ + "#### Import Vertex client library\n", + "\n", + "Import the Vertex client library into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:protobuf" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `API_ENDPOINT`: The Vertex API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex location root path for dataset, model, job, pipeline and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API service endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accelerators:prediction,cpu" + }, + "source": [ + "#### Hardware Accelerators\n", + "\n", + "Set the hardware accelerators (e.g., GPU), if any, for prediction.\n", + "\n", + "Set the variable `DEPLOY_GPU/DEPLOY_NGPU` to use a container image supporting a GPU and the number of GPUs allocated to the virtual machine (VM) instance. For example, to use a GPU container image with 4 Nvidia Telsa K80 GPUs allocated to each VM, you would specify:\n", + "\n", + " (aip.AcceleratorType.NVIDIA_TESLA_K80, 4)\n", + "\n", + "For GPU, available accelerators include:\n", + " - aip.AcceleratorType.NVIDIA_TESLA_K80\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P100\n", + " - aip.AcceleratorType.NVIDIA_TESLA_P4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_T4\n", + " - aip.AcceleratorType.NVIDIA_TESLA_V100\n", + "\n", + "Otherwise specify `(None, None)` to use a container image to run on a CPU." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accelerators:prediction,cpu" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPOLY_GPU\"):\n", + " DEPLOY_GPU, DEPLOY_NGPU = (\n", + " aip.AcceleratorType.NVIDIA_TESLA_K80,\n", + " int(os.getenv(\"IS_TESTING_DEPOLY_GPU\")),\n", + " )\n", + "else:\n", + " DEPLOY_GPU, DEPLOY_NGPU = (None, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "container:prediction" + }, + "source": [ + "#### Container (Docker) image\n", + "\n", + "Next, we will set the Docker container images for prediction\n", + "\n", + "- Set the variable `TF` to the TensorFlow version of the container image. For example, `2-1` would be version 2.1, and `1-15` would be version 1.15. The following list shows some of the pre-built images available:\n", + "\n", + " - TensorFlow 1.15\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-cpu.1-15:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf-gpu.1-15:latest`\n", + " - TensorFlow 2.1\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-1:latest`\n", + " - TensorFlow 2.2\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-2:latest`\n", + " - TensorFlow 2.3\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-3:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-3:latest`\n", + " - XGBoost\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-2:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-1:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-90:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/xgboost-cpu.0-82:latest`\n", + " - Scikit-learn\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-23:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-22:latest`\n", + " - `gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-20:latest`\n", + "\n", + "For the latest list, see [Pre-built containers for prediction](https://cloud.google.com/vertex-ai/docs/predictions/pre-built-containers)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "container:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_TF\"):\n", + " TF = os.getenv(\"IS_TESTING_TF\")\n", + "else:\n", + " TF = \"2-1\"\n", + "\n", + "if TF[0] == \"2\":\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf2-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf2-cpu.{}\".format(TF)\n", + "else:\n", + " if DEPLOY_GPU:\n", + " DEPLOY_VERSION = \"tf-gpu.{}\".format(TF)\n", + " else:\n", + " DEPLOY_VERSION = \"tf-cpu.{}\".format(TF)\n", + "\n", + "DEPLOY_IMAGE = \"gcr.io/cloud-aiplatform/prediction/{}:latest\".format(DEPLOY_VERSION)\n", + "\n", + "print(\"Deployment:\", DEPLOY_IMAGE, DEPLOY_GPU)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "machine:prediction" + }, + "source": [ + "#### Machine Type\n", + "\n", + "Next, set the machine type to use for prediction.\n", + "\n", + "- Set the variable `DEPLOY_COMPUTE` to configure the compute resources for the VM you will use for prediction.\n", + " - `machine type`\n", + " - `n1-standard`: 3.75GB of memory per vCPU.\n", + " - `n1-highmem`: 6.5GB of memory per vCPU\n", + " - `n1-highcpu`: 0.9 GB of memory per vCPU\n", + " - `vCPUs`: number of \\[2, 4, 8, 16, 32, 64, 96 \\]\n", + "\n", + "*Note: You may also use n2 and e2 machine types for training and deployment, but they do not support GPUs*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "machine:prediction" + }, + "outputs": [], + "source": [ + "if os.getenv(\"IS_TESTING_DEPLOY_MACHINE\"):\n", + " MACHINE_TYPE = os.getenv(\"IS_TESTING_DEPLOY_MACHINE\")\n", + "else:\n", + " MACHINE_TYPE = \"n1-standard\"\n", + "\n", + "VCPU = \"4\"\n", + "DEPLOY_COMPUTE = MACHINE_TYPE + \"-\" + VCPU\n", + "print(\"Deploy machine type\", DEPLOY_COMPUTE)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:tfhub" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to deploy a TensorFlow Hub pretrained image classification model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients:tfhub" + }, + "source": [ + "## Set up clients\n", + "\n", + "The Vertex client library works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the Vertex server.\n", + "\n", + "You will use different clients in this tutorial for different steps in the workflow. So set them all up upfront.\n", + "\n", + "- Model Service for `Model` resources.\n", + "- Endpoint Service for deployment.\n", + "- Prediction Service for serving." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients:tfhub" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_tf_hub_model" + }, + "source": [ + "## Get pretrained model from TFHub\n", + "\n", + "Next, you download a pre-trained model from $(TENSORFLOW) Hub." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_tf_hub_model" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "import tensorflow_hub as hub\n", + "\n", + "IMAGE_SHAPE = (224, 224)\n", + "\n", + "model = tf.keras.Sequential(\n", + " [\n", + " hub.KerasLayer(\n", + " \"https://tfhub.dev/google/imagenet/resnet_v1_50/feature_vector/4\",\n", + " input_shape=IMAGE_SHAPE + (3,),\n", + " )\n", + " ]\n", + ")\n", + "\n", + "model_path_to_deploy = BUCKET_NAME + \"/resnet\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "how_serving_function_works" + }, + "source": [ + "## Upload the model for serving\n", + "\n", + "Next, you will upload your TF.Keras model from the custom job to Vertex `Model` service, which will create a Vertex `Model` resource for your custom model. During upload, you need to define a serving function to convert data to the format your model expects. If you send encoded data to Vertex, your serving function ensures that the data is decoded on the model server before it is passed as input to your model.\n", + "\n", + "### How does the serving function work\n", + "\n", + "When you send a request to an online prediction server, the request is received by a HTTP server. The HTTP server extracts the prediction request from the HTTP request content body. The extracted prediction request is forwarded to the serving function. For Google pre-built prediction containers, the request content is passed to the serving function as a `tf.string`.\n", + "\n", + "The serving function consists of two parts:\n", + "\n", + "- `preprocessing function`:\n", + " - Converts the input (`tf.string`) to the input shape and data type of the underlying model (dynamic graph).\n", + " - Performs the same preprocessing of the data that was done during training the underlying model -- e.g., normalizing, scaling, etc.\n", + "- `post-processing function`:\n", + " - Converts the model output to format expected by the receiving application -- e.q., compresses the output.\n", + " - Packages the output for the the receiving application -- e.g., add headings, make JSON object, etc.\n", + "\n", + "Both the preprocessing and post-processing functions are converted to static graphs which are fused to the model. The output from the underlying model is passed to the post-processing function. The post-processing function passes the converted/packaged output back to the HTTP server. The HTTP server returns the output as the HTTP response content.\n", + "\n", + "One consideration you need to consider when building serving functions for TF.Keras models is that they run as static graphs. That means, you cannot use TF graph operations that require a dynamic graph. If you do, you will get an error during the compile of the serving function which will indicate that you are using an EagerTensor which is not supported." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_image" + }, + "source": [ + "### Serving function for image data\n", + "\n", + "To pass images to the prediction service, you encode the compressed (e.g., JPEG) image bytes into base 64 -- which makes the content safe from modification while transmitting binary data over the network. Since this deployed model expects input data as raw (uncompressed) bytes, you need to ensure that the base 64 encoded data gets converted back to raw bytes before it is passed as input to the deployed model.\n", + "\n", + "To resolve this, define a serving function (`serving_fn`) and attach it to the model as a preprocessing step. Add a `@tf.function` decorator so the serving function is fused to the underlying model (instead of upstream on a CPU).\n", + "\n", + "When you send a prediction or explanation request, the content of the request is base 64 decoded into a Tensorflow string (`tf.string`), which is passed to the serving function (`serving_fn`). The serving function preprocesses the `tf.string` into raw (uncompressed) numpy bytes (`preprocess_fn`) to match the input requirements of the model:\n", + "- `io.decode_jpeg`- Decompresses the JPG image which is returned as a Tensorflow tensor with three channels (RGB).\n", + "- `image.convert_image_dtype` - Changes integer pixel values to float 32.\n", + "- `image.resize` - Resizes the image to match the input shape for the model.\n", + "- `resized / 255.0` - Rescales (normalization) the pixel data between 0 and 1.\n", + "\n", + "At this point, the data can be passed to the model (`m_call`)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_image" + }, + "outputs": [], + "source": [ + "CONCRETE_INPUT = \"numpy_inputs\"\n", + "\n", + "\n", + "def _preprocess(bytes_input):\n", + " decoded = tf.io.decode_jpeg(bytes_input, channels=3)\n", + " decoded = tf.image.convert_image_dtype(decoded, tf.float32)\n", + " resized = tf.image.resize(decoded, size=(32, 32))\n", + " rescale = tf.cast(resized / 255.0, tf.float32)\n", + " return rescale\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def preprocess_fn(bytes_inputs):\n", + " decoded_images = tf.map_fn(\n", + " _preprocess, bytes_inputs, dtype=tf.float32, back_prop=False\n", + " )\n", + " return {\n", + " CONCRETE_INPUT: decoded_images\n", + " } # User needs to make sure the key matches model's input\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def serving_fn(bytes_inputs):\n", + " images = preprocess_fn(bytes_inputs)\n", + " prob = m_call(**images)\n", + " return prob\n", + "\n", + "\n", + "m_call = tf.function(model.call).get_concrete_function(\n", + " [tf.TensorSpec(shape=[None, 32, 32, 3], dtype=tf.float32, name=CONCRETE_INPUT)]\n", + ")\n", + "\n", + "tf.saved_model.save(\n", + " model, model_path_to_deploy, signatures={\"serving_default\": serving_fn}\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "serving_function_signature:image" + }, + "source": [ + "## Get the serving function signature\n", + "\n", + "You can get the signatures of your model's input and output layers by reloading the model into memory, and querying it for the signatures corresponding to each layer.\n", + "\n", + "For your purpose, you need the signature of the serving function. Why? Well, when we send our data for prediction as a HTTP request packet, the image data is base64 encoded, and our TF.Keras model takes numpy input. Your serving function will do the conversion from base64 to a numpy array.\n", + "\n", + "When making a prediction request, you need to route the request to the serving function instead of the model, so you need to know the input layer name of the serving function -- which you will use later when you make a prediction request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "serving_function_signature:image" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_path_to_deploy)\n", + "\n", + "serving_input = list(\n", + " loaded.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "print(\"Serving function input:\", serving_input)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "upload_the_model" + }, + "source": [ + "### Upload the model\n", + "\n", + "Use this helper function `upload_model` to upload your model, stored in SavedModel format, up to the `Model` service, which will instantiate a Vertex `Model` resource instance for your model. Once you've done that, you can use the `Model` resource instance in the same way as any other Vertex `Model` resource instance, such as deploying to an `Endpoint` resource for serving predictions.\n", + "\n", + "The helper function takes the following parameters:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` service.\n", + "- `image_uri`: The container image for the model deployment.\n", + "- `model_uri`: The Cloud Storage path to our SavedModel artificat. For this tutorial, this is the Cloud Storage location where the `trainer/task.py` saved the model artifacts, which we specified in the variable `MODEL_DIR`.\n", + "\n", + "The helper function calls the `Model` client service's method `upload_model`, which takes the following parameters:\n", + "\n", + "- `parent`: The Vertex location root path for `Dataset`, `Model` and `Endpoint` resources.\n", + "- `model`: The specification for the Vertex `Model` resource instance.\n", + "\n", + "Let's now dive deeper into the Vertex model specification `model`. This is a dictionary object that consists of the following fields:\n", + "\n", + "- `display_name`: A human readable name for the `Model` resource.\n", + "- `metadata_schema_uri`: Since your model was built without an Vertex `Dataset` resource, you will leave this blank (`''`).\n", + "- `artificat_uri`: The Cloud Storage path where the model is stored in SavedModel format.\n", + "- `container_spec`: This is the specification for the Docker container that will be installed on the `Endpoint` resource, from which the `Model` resource will serve predictions. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + "\n", + "Uploading a model into a Vertex Model resource returns a long running operation, since it may take a few moments. You call response.result(), which is a synchronous call and will return when the Vertex Model resource is ready.\n", + "\n", + "The helper function returns the Vertex fully qualified identifier for the corresponding Vertex Model instance upload_model_response.model. You will save the identifier for subsequent steps in the variable model_to_deploy_id." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "upload_the_model" + }, + "outputs": [], + "source": [ + "IMAGE_URI = DEPLOY_IMAGE\n", + "\n", + "\n", + "def upload_model(display_name, image_uri, model_uri):\n", + " model = {\n", + " \"display_name\": display_name,\n", + " \"metadata_schema_uri\": \"\",\n", + " \"artifact_uri\": model_uri,\n", + " \"container_spec\": {\n", + " \"image_uri\": image_uri,\n", + " \"command\": [],\n", + " \"args\": [],\n", + " \"env\": [{\"name\": \"env_name\", \"value\": \"env_value\"}],\n", + " \"ports\": [{\"container_port\": 8080}],\n", + " \"predict_route\": \"\",\n", + " \"health_route\": \"\",\n", + " },\n", + " }\n", + " response = clients[\"model\"].upload_model(parent=PARENT, model=model)\n", + " print(\"Long running operation:\", response.operation.name)\n", + " upload_model_response = response.result(timeout=180)\n", + " print(\"upload_model_response\")\n", + " print(\" model:\", upload_model_response.model)\n", + " return upload_model_response.model\n", + "\n", + "\n", + "model_to_deploy_id = upload_model(\n", + " \"flowers-\" + TIMESTAMP, IMAGE_URI, model_path_to_deploy\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_model" + }, + "source": [ + "### Get `Model` resource information\n", + "\n", + "Now let's get the model information for just your model. Use this helper function `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource.\n", + "\n", + "This helper function calls the Vertex `Model` client service's method `get_model`, with the following parameter:\n", + "\n", + "- `name`: The Vertex unique identifier for the `Model` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_model" + }, + "outputs": [], + "source": [ + "def get_model(name):\n", + " response = clients[\"model\"].get_model(name=name)\n", + " print(response)\n", + "\n", + "\n", + "get_model(model_to_deploy_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint:custom" + }, + "source": [ + "## Deploy the `Model` resource\n", + "\n", + "Now deploy the trained Vertex custom `Model` resource. This requires two steps:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_endpoint" + }, + "source": [ + "### Create an `Endpoint` resource\n", + "\n", + "Use this helper function `create_endpoint` to create an endpoint to deploy the model to for serving predictions, with the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "The helper function uses the endpoint client service's `create_endpoint` method, which takes the following parameter:\n", + "\n", + "- `display_name`: A human readable name for the `Endpoint` resource.\n", + "\n", + "Creating an `Endpoint` resource returns a long running operation, since it may take a few moments to provision the `Endpoint` resource for serving. You call `response.result()`, which is a synchronous call and will return when the Endpoint resource is ready. The helper function returns the Vertex fully qualified identifier for the `Endpoint` resource: `response.name`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_endpoint" + }, + "outputs": [], + "source": [ + "ENDPOINT_NAME = \"flowers_endpoint-\" + TIMESTAMP\n", + "\n", + "\n", + "def create_endpoint(display_name):\n", + " endpoint = {\"display_name\": display_name}\n", + " response = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)\n", + " print(\"Long running operation:\", response.operation.name)\n", + "\n", + " result = response.result(timeout=300)\n", + " print(\"result\")\n", + " print(\" name:\", result.name)\n", + " print(\" display_name:\", result.display_name)\n", + " print(\" description:\", result.description)\n", + " print(\" labels:\", result.labels)\n", + " print(\" create_time:\", result.create_time)\n", + " print(\" update_time:\", result.update_time)\n", + " return result\n", + "\n", + "\n", + "result = create_endpoint(ENDPOINT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoint_id:result" + }, + "source": [ + "Now get the unique identifier for the `Endpoint` resource you created." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:result" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "instance_scaling" + }, + "source": [ + "### Compute instance scaling\n", + "\n", + "You have several choices on scaling the compute instances for handling your online prediction requests:\n", + "\n", + "- Single Instance: The online prediction requests are processed on a single compute instance.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to one.\n", + "\n", + "- Manual Scaling: The online prediction requests are split across a fixed number of compute instances that you manually specified.\n", + " - Set the minimum (`MIN_NODES`) and maximum (`MAX_NODES`) number of compute instances to the same number of nodes. When a model is first deployed to the instance, the fixed number of compute instances are provisioned and online prediction requests are evenly distributed across them.\n", + "\n", + "- Auto Scaling: The online prediction requests are split across a scaleable number of compute instances.\n", + " - Set the minimum (`MIN_NODES`) number of compute instances to provision when a model is first deployed and to de-provision, and set the maximum (`MAX_NODES) number of compute instances to provision, depending on load conditions.\n", + "\n", + "The minimum number of compute instances corresponds to the field `min_replica_count` and the maximum number of compute instances corresponds to the field `max_replica_count`, in your subsequent deployment request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instance_scaling" + }, + "outputs": [], + "source": [ + "MIN_NODES = 1\n", + "MAX_NODES = 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:dedicated" + }, + "source": [ + "### Deploy `Model` resource to the `Endpoint` resource\n", + "\n", + "Use this helper function `deploy_model` to deploy the `Model` resource to the `Endpoint` resource you created for serving predictions, with the following parameters:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the model to upload (deploy) from the training pipeline.\n", + "- `deploy_model_display_name`: A human readable name for the deployed model.\n", + "- `endpoint`: The Vertex fully qualified endpoint identifier to deploy the model to.\n", + "\n", + "The helper function calls the `Endpoint` client service's method `deploy_model`, which takes the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified `Endpoint` resource identifier to deploy the `Model` resource to.\n", + "- `deployed_model`: The requirements specification for deploying the model.\n", + "- `traffic_split`: Percent of traffic at the endpoint that goes to this model, which is specified as a dictionary of one or more key/value pairs.\n", + " - If only one model, then specify as **{ \"0\": 100 }**, where \"0\" refers to this model being uploaded and 100 means 100% of the traffic.\n", + " - If there are existing models on the endpoint, for which the traffic will be split, then use `model_id` to specify as **{ \"0\": percent, model_id: percent, ... }**, where `model_id` is the model id of an existing model to the deployed endpoint. The percents must add up to 100.\n", + "\n", + "Let's now dive deeper into the `deployed_model` parameter. This parameter is specified as a Python dictionary with the minimum required fields:\n", + "\n", + "- `model`: The Vertex fully qualified model identifier of the (upload) model to deploy.\n", + "- `display_name`: A human readable name for the deployed model.\n", + "- `disable_container_logging`: This disables logging of container events, such as execution failures (default is container logging is enabled). Container logging is typically enabled when debugging the deployment and then disabled when deployed for production.\n", + "- `dedicated_resources`: This refers to how many compute instances (replicas) that are scaled for serving prediction requests.\n", + " - `machine_spec`: The compute instance to provision. Use the variable you set earlier `DEPLOY_GPU != None` to use a GPU; otherwise only a CPU is allocated.\n", + " - `min_replica_count`: The number of compute instances to initially provision, which you set earlier as the variable `MIN_NODES`.\n", + " - `max_replica_count`: The maximum number of compute instances to scale to, which you set earlier as the variable `MAX_NODES`.\n", + "\n", + "#### Traffic Split\n", + "\n", + "Let's now dive deeper into the `traffic_split` parameter. This parameter is specified as a Python dictionary. This might at first be a tad bit confusing. Let me explain, you can deploy more than one instance of your model to an endpoint, and then set how much (percent) goes to each instance.\n", + "\n", + "Why would you do that? Perhaps you already have a previous version deployed in production -- let's call that v1. You got better model evaluation on v2, but you don't know for certain that it is really better until you deploy to production. So in the case of traffic split, you might want to deploy v2 to the same endpoint as v1, but it only get's say 10% of the traffic. That way, you can monitor how well it does without disrupting the majority of users -- until you make a final decision.\n", + "\n", + "#### Response\n", + "\n", + "The method returns a long running operation `response`. We will wait sychronously for the operation to complete by calling the `response.result()`, which will block until the model is deployed. If this is the first time a model is deployed to the endpoint, it may take a few additional minutes to complete provisioning of resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:dedicated" + }, + "outputs": [], + "source": [ + "DEPLOYED_NAME = \"flowers_deployed-\" + TIMESTAMP\n", + "\n", + "\n", + "def deploy_model(\n", + " model, deployed_model_display_name, endpoint, traffic_split={\"0\": 100}\n", + "):\n", + "\n", + " if DEPLOY_GPU:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_type\": DEPLOY_GPU,\n", + " \"accelerator_count\": DEPLOY_NGPU,\n", + " }\n", + " else:\n", + " machine_spec = {\n", + " \"machine_type\": DEPLOY_COMPUTE,\n", + " \"accelerator_count\": 0,\n", + " }\n", + "\n", + " deployed_model = {\n", + " \"model\": model,\n", + " \"display_name\": deployed_model_display_name,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": MIN_NODES,\n", + " \"max_replica_count\": MAX_NODES,\n", + " \"machine_spec\": machine_spec,\n", + " },\n", + " \"disable_container_logging\": False,\n", + " }\n", + "\n", + " response = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint, deployed_model=deployed_model, traffic_split=traffic_split\n", + " )\n", + "\n", + " print(\"Long running operation:\", response.operation.name)\n", + " result = response.result()\n", + " print(\"result\")\n", + " deployed_model = result.deployed_model\n", + " print(\" deployed_model\")\n", + " print(\" id:\", deployed_model.id)\n", + " print(\" model:\", deployed_model.model)\n", + " print(\" display_name:\", deployed_model.display_name)\n", + " print(\" create_time:\", deployed_model.create_time)\n", + "\n", + " return deployed_model.id\n", + "\n", + "\n", + "deployed_model_id = deploy_model(model_to_deploy_id, DEPLOYED_NAME, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_item:image" + }, + "source": [ + "### Get test item\n", + "\n", + "You will use an example image from your dataset as a test item." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:image" + }, + "outputs": [], + "source": [ + "FLOWERS_CSV = \"gs://cloud-ml-data/img/flower_photos/all_data.csv\"\n", + "\n", + "test_images = ! gsutil cat $FLOWERS_CSV | head -n1\n", + "test_image = test_images[0].split(\",\")[0]\n", + "print(test_image)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "prepare_test_item:image" + }, + "source": [ + "### Prepare the request content\n", + "\n", + "You are going to send the flowers image as compressed JPG image, instead of the raw uncompressed bytes:\n", + "\n", + "- `tf.io.read_file`: Read the compressed JPG images back into memory as raw bytes.\n", + "- `base64.b64encode`: Encode the raw bytes into a base 64 encoded string." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "prepare_test_item:image" + }, + "outputs": [], + "source": [ + "import base64\n", + "\n", + "bytes = tf.io.read_file(test_image)\n", + "b64str = base64.b64encode(bytes.numpy()).decode(\"utf-8\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "send_prediction_request:image" + }, + "source": [ + "### Send the prediction request\n", + "\n", + "Ok, now you have a test image. Use this helper function `predict_image`, which takes the following parameters:\n", + "\n", + "- `image`: The test image data as a numpy array.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed to.\n", + "- `parameters_dict`: Additional parameters for serving.\n", + "\n", + "This function calls the prediction client service `predict` method with the following parameters:\n", + "\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource was deployed to.\n", + "- `instances`: A list of instances (encoded images) to predict.\n", + "- `parameters`: Additional parameters for serving.\n", + "\n", + "To pass the image data to the prediction service, in the previous step you encoded the bytes into base64 -- which makes the content safe from modification when transmitting binary data over the network. You need to tell the serving binary where your model is deployed to, that the content has been base64 encoded, so it will decode it on the other end in the serving binary.\n", + "\n", + "Each instance in the prediction request is a dictionary entry of the form:\n", + "\n", + " {serving_input: {'b64': content}}\n", + "\n", + "- `input_name`: the name of the input layer of the underlying model.\n", + "- `'b64'`: A key that indicates the content is base64 encoded.\n", + "- `content`: The compressed JPG image bytes as a base64 encoded string.\n", + "\n", + "Since the `predict()` service can take multiple images (instances), you will send your single image as a list of one image. As a final step, you package the instances list into Google's protobuf format -- which is what we pass to the `predict()` service.\n", + "\n", + "The `response` object returns a list, where each element in the list corresponds to the corresponding image in the request. You will see in the output for each prediction:\n", + "\n", + "- `predictions`: Confidence level for the prediction, between 0 and 1, for each of the classes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "send_prediction_request:image" + }, + "outputs": [], + "source": [ + "def predict_image(image, endpoint, parameters_dict):\n", + " # The format of each instance should conform to the deployed model's prediction input schema.\n", + " instances_list = [{serving_input: {\"b64\": image}}]\n", + " instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + " response = clients[\"prediction\"].predict(\n", + " endpoint=endpoint, instances=instances, parameters=parameters_dict\n", + " )\n", + " print(\"response\")\n", + " print(\" deployed_model_id:\", response.deployed_model_id)\n", + " predictions = response.predictions\n", + " print(\"predictions\")\n", + " for prediction in predictions:\n", + " print(\" prediction:\", prediction)\n", + "\n", + "\n", + "predict_image(b64str, endpoint_id, None)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model" + }, + "source": [ + "## Undeploy the `Model` resource\n", + "\n", + "Now undeploy your `Model` resource from the serving `Endpoint` resoure. Use this helper function `undeploy_model`, which takes the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed to.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` is deployed to.\n", + "\n", + "This function calls the endpoint client service's method `undeploy_model`, with the following parameters:\n", + "\n", + "- `deployed_model_id`: The model deployment identifier returned by the endpoint service when the `Model` resource was deployed.\n", + "- `endpoint`: The Vertex fully qualified identifier for the `Endpoint` resource where the `Model` resource is deployed.\n", + "- `traffic_split`: How to split traffic among the remaining deployed models on the `Endpoint` resource.\n", + "\n", + "Since this is the only deployed model on the `Endpoint` resource, you simply can leave `traffic_split` empty by setting it to {}." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model" + }, + "outputs": [], + "source": [ + "def undeploy_model(deployed_model_id, endpoint):\n", + " response = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint, deployed_model_id=deployed_model_id, traffic_split={}\n", + " )\n", + " print(response)\n", + "\n", + "\n", + "undeploy_model(deployed_model_id, endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset and \"dataset_id\" in globals():\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex fully qualified identifier for the pipeline\n", + "try:\n", + " if delete_pipeline and \"pipeline_id\" in globals():\n", + " clients[\"pipeline\"].delete_training_pipeline(name=pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model and \"model_to_deploy_id\" in globals():\n", + " clients[\"model\"].delete_model(name=model_to_deploy_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint and \"endpoint_id\" in globals():\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob and \"batch_job_id\" in globals():\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom job using the Vertex fully qualified identifier for the custom job\n", + "try:\n", + " if delete_customjob and \"job_id\" in globals():\n", + " clients[\"job\"].delete_custom_job(name=job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the hyperparameter tuning job using the Vertex fully qualified identifier for the hyperparameter tuning job\n", + "try:\n", + " if delete_hptjob and \"hpt_job_id\" in globals():\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hpt_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "showcase_tfhub_image_classification_online.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/hyperparameter_tuning/distributed-hyperparameter-tuning.ipynb b/notebooks/community/hyperparameter_tuning/distributed-hyperparameter-tuning.ipynb new file mode 100644 index 000000000..ab8127536 --- /dev/null +++ b/notebooks/community/hyperparameter_tuning/distributed-hyperparameter-tuning.ipynb @@ -0,0 +1,889 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ur8xi4C7S06n" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "JAPoU8Sm5E6e" + }, + "source": [ + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tvgnzT1CKxrO" + }, + "source": [ + "## Overview\n", + "\n", + "This notebook demonstrates how to run a hyperparameter tuning job with Vertex Training to discover optimal hyperparameter values for an ML model. To speed up the training process, `MirroredStrategy` from the `tf.distribute` module is used to distribute training across multiple GPUs on a single machine.\n", + "\n", + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [horses or humans dataset](https://www.tensorflow.org/datasets/catalog/horses_or_humans) from [TensorFlow Datasets](https://www.tensorflow.org/datasets). The trained model predicts if an image is of a horse or a human.\n", + "\n", + "### Objective\n", + "\n", + "In this notebook, you create a custom-trained model from a Python script in a Docker container. You learn how to modify training application code for hyperparameter tuning and submit a Vertex Training hyperparameter tuning job with the Python SDK.\n", + "\n", + "The steps performed include:\n", + "\n", + "* Create a Vertex AI custom job for training a model.\n", + "* Launch hyperparameter tuning job with the Python SDK.\n", + "* Cleanup resources.\n", + "\n", + "\n", + "### Costs \n", + "\n", + "This tutorial uses billable components of Google Cloud:\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ze4-nDLfK4pw" + }, + "source": [ + "### Set up your local development environment\n", + "\n", + "**If you are using Colab or Google Cloud Notebooks**, your environment already meets\n", + "all the requirements to run this notebook. You can skip this step." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gCuSR8GkAgzl" + }, + "source": [ + "**Otherwise**, make sure your environment meets this notebook's requirements.\n", + "You need the following:\n", + "\n", + "* The Google Cloud SDK\n", + "* Git\n", + "* Python 3\n", + "* virtualenv\n", + "* Jupyter notebook running in a virtual environment with Python 3\n", + "\n", + "The Google Cloud guide to [Setting up a Python development\n", + "environment](https://cloud.google.com/python/setup) and the [Jupyter\n", + "installation guide](https://jupyter.org/install) provide detailed instructions\n", + "for meeting these requirements. The following steps provide a condensed set of\n", + "instructions:\n", + "\n", + "1. [Install and initialize the Cloud SDK.](https://cloud.google.com/sdk/docs/)\n", + "\n", + "1. [Install Python 3.](https://cloud.google.com/python/setup#installing_python)\n", + "\n", + "1. [Install\n", + " virtualenv](https://cloud.google.com/python/setup#installing_and_using_virtualenv)\n", + " and create a virtual environment that uses Python 3. Activate the virtual environment.\n", + "\n", + "1. To install Jupyter, run `pip3 install jupyter` on the\n", + "command-line in a terminal shell.\n", + "\n", + "1. To launch Jupyter, run `jupyter notebook` on the command-line in a terminal shell.\n", + "\n", + "1. Open this notebook in the Jupyter Notebook Dashboard." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "i7EUnXsZhAGF" + }, + "source": [ + "### Install additional packages\n", + "\n", + "Install the latest version of Vertex SDK for Python." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "2b4ef9b72d43" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "# The Google Cloud Notebook product has specific requirements\n", + "IS_GOOGLE_CLOUD_NOTEBOOK = os.path.exists(\"/opt/deeplearning/metadata/env_version\")\n", + "\n", + "# Google Cloud Notebook requires dependencies to be installed with '--user'\n", + "USER_FLAG = \"\"\n", + "if IS_GOOGLE_CLOUD_NOTEBOOK:\n", + " USER_FLAG = \"--user\"" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wyy5Lbnzg5fi" + }, + "outputs": [], + "source": [ + "! pip3 install {USER_FLAG} --upgrade google-cloud-aiplatform" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "hhq5zEbGg0XX" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "After you install the additional packages, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "EzrelQZ22IZj" + }, + "outputs": [], + "source": [ + "# Automatically restart kernel after installs\n", + "import os\n", + "\n", + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "lWEdiXsJg0XY" + }, + "source": [ + "## Before you begin\n", + "\n", + "### Select a GPU runtime\n", + "\n", + "**Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select \"Runtime --> Change runtime type > GPU\"**" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "BF1j6f9HApxa" + }, + "source": [ + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n", + "\n", + "1. [Enable the Vertex AI API and Compute Engine API](https://console.cloud.google.com/flows/enableapi?apiid=aiplatform.googleapis.com,compute_component).\n", + "1. If you are running this notebook locally, you will need to install the [Cloud SDK](https://cloud.google.com/sdk).\n", + "\n", + "1. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "WReHDGG5g0XY" + }, + "source": [ + "#### Set your project ID\n", + "\n", + "**If you don't know your project ID**, you may be able to get your project ID using `gcloud`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "oM1iC_MfAts1" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"\"\n", + "\n", + "# Get your Google Cloud project ID from gcloud\n", + "if not os.getenv(\"IS_TESTING\"):\n", + " shell_output=!gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID: \", PROJECT_ID)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "qJYoRfYng0XZ" + }, + "source": [ + "Otherwise, set your project ID here." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "riG_qUokg0XZ" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None:\n", + " PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "06571eb4063b" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append it onto the name of resources you create in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "697568e92bd6" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dr--iN2kAylZ" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebooks**, your environment is already\n", + "authenticated. Skip this step." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sBCra4QMA2wR" + }, + "source": [ + "**If you are using Colab**, run the cell below and follow the instructions\n", + "when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "1. In the Cloud Console, go to the [**Create service account key**\n", + " page](https://console.cloud.google.com/apis/credentials/serviceaccountkey).\n", + "\n", + "2. Click **Create service account**.\n", + "\n", + "3. In the **Service account name** field, enter a name, and\n", + " click **Create**.\n", + "\n", + "4. In the **Grant this service account access to project** section, click the **Role** drop-down list. Type \"Vertex AI\"\n", + "into the filter box, and select\n", + " **Vertex AI Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "5. Click *Create*. A JSON file that contains your key downloads to your\n", + "local environment.\n", + "\n", + "6. Enter the path to your service account key as the\n", + "`GOOGLE_APPLICATION_CREDENTIALS` variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "PyQmSRbKA8r-" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# The Google Cloud Notebook product has specific requirements\n", + "IS_GOOGLE_CLOUD_NOTEBOOK = os.path.exists(\"/opt/deeplearning/metadata/env_version\")\n", + "\n", + "# If on Google Cloud Notebooks, then don't execute this code\n", + "if not IS_GOOGLE_CLOUD_NOTEBOOK:\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "zgPO1eR3CYjk" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you submit a custom training job using the Cloud SDK, you will need to provide a staging bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all\n", + "Cloud Storage buckets.\n", + "\n", + "You may also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Make sure to [choose a region where Vertex AI services are\n", + "available](https://cloud.google.com/vertex-ai/docs/general/locations#available_regions). You may\n", + "not use a Multi-Regional Storage bucket for training with Vertex AI." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "MzGDU7TWdts_" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}\n", + "REGION = \"[your-region]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cf221059d072" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "-EcIXiGsCePi" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "NIq7R4HZCfIc" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ucvCsknMCims" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "vhOb7YnwClBb" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "XoEqT2Y4DJmf" + }, + "source": [ + "### Import libraries and define constants" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pRUOFELefqf1" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "from google.cloud import aiplatform\n", + "from google.cloud.aiplatform import hyperparameter_tuning as hpt" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "reBCSTKOg47l" + }, + "source": [ + "### Write Dockerfile\n", + "\n", + "The first step in containerizing your code is to create a Dockerfile. In the Dockerfile, you'll include all the commands needed to run the image such as installing the necessary libraries and setting up the entry point for the training code.\n", + "\n", + "This Dockerfile uses the Deep Learning Container TensorFlow Enterprise 2.5 GPU Docker image. The Deep Learning Containers on Google Cloud come with many common ML and data science frameworks pre-installed. After downloading that image, this Dockerfile installs the [CloudML Hypertune](https://github.com/GoogleCloudPlatform/cloudml-hypertune) library and sets up the entrypoint for the training code.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "%%writefile Dockerfile\n", + "\n", + "FROM gcr.io/deeplearning-platform-release/tf2-gpu.2-5\n", + "WORKDIR /\n", + "\n", + "# Installs hypertune library\n", + "RUN pip install cloudml-hypertune\n", + "\n", + "# Copies the trainer code to the docker image.\n", + "COPY trainer /trainer\n", + "\n", + "# Sets up the entry point to invoke the trainer.\n", + "ENTRYPOINT [\"python\", \"-m\", \"trainer.task\"]" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### Create training application code\n", + "\n", + "Next, you create a trainer directory with a `task.py` script that contains the code for your training application." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "MjJTYC86hPOZ" + }, + "outputs": [], + "source": [ + "# Create trainer directory\n", + "\n", + "! mkdir trainer" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "In the next cell, you write the contents of the training script, `task.py`. This file downloads the _horses or humans_ dataset from TensorFlow datasets and trains a `tf.keras` functional model using `MirroredStrategy` from the `tf.distribute` module.\n", + "\n", + "There are a few components that are specific to using the hyperparameter tuning service:\n", + "\n", + "* The script imports the `hypertune` library. Note that the Dockerfile included instructions to pip install the hypertune library.\n", + "* The function `get_args()` defines a command-line argument for each hyperparameter you want to tune. In this example, the hyperparameters that will be tuned are the learning rate, the momentum value in the optimizer, and the number of units in the last hidden layer of the model. The value passed in those arguments is then used to set the corresponding hyperparameter in the code.\n", + "* At the end of the `main()` function, the hypertune library is used to define the metric to optimize. In this example, the metric that will be optimized is the the validation accuracy. This metric is passed to an instance of `HyperTune`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "%%writefile trainer/task.py\n", + "\n", + "import argparse\n", + "import hypertune\n", + "import tensorflow as tf\n", + "import tensorflow_datasets as tfds\n", + "\n", + "def get_args():\n", + " \"\"\"Parses args. Must include all hyperparameters you want to tune.\"\"\"\n", + "\n", + " parser = argparse.ArgumentParser()\n", + " parser.add_argument(\n", + " '--learning_rate', required=True, type=float, help='learning rate')\n", + " parser.add_argument(\n", + " '--momentum', required=True, type=float, help='SGD momentum value')\n", + " parser.add_argument(\n", + " '--units',\n", + " required=True,\n", + " type=int,\n", + " help='number of units in last hidden layer')\n", + " parser.add_argument(\n", + " '--epochs',\n", + " required=False,\n", + " type=int,\n", + " default=10,\n", + " help='number of training epochs')\n", + " args = parser.parse_args()\n", + " return args\n", + "\n", + "\n", + "def preprocess_data(image, label):\n", + " \"\"\"Resizes and scales images.\"\"\"\n", + "\n", + " image = tf.image.resize(image, (150, 150))\n", + " return tf.cast(image, tf.float32) / 255., label\n", + "\n", + "\n", + "def create_dataset(batch_size):\n", + " \"\"\"Loads Horses Or Humans dataset and preprocesses data.\"\"\"\n", + "\n", + " data, info = tfds.load(\n", + " name='horses_or_humans', as_supervised=True, with_info=True)\n", + "\n", + " # Create train dataset\n", + " train_data = data['train'].map(preprocess_data)\n", + " train_data = train_data.shuffle(1000)\n", + " train_data = train_data.batch(batch_size)\n", + "\n", + " # Create validation dataset\n", + " validation_data = data['test'].map(preprocess_data)\n", + " validation_data = validation_data.batch(64)\n", + "\n", + " return train_data, validation_data\n", + "\n", + "\n", + "def create_model(units, learning_rate, momentum):\n", + " \"\"\"Defines and compiles model.\"\"\"\n", + "\n", + " inputs = tf.keras.Input(shape=(150, 150, 3))\n", + " x = tf.keras.layers.Conv2D(16, (3, 3), activation='relu')(inputs)\n", + " x = tf.keras.layers.MaxPooling2D((2, 2))(x)\n", + " x = tf.keras.layers.Conv2D(32, (3, 3), activation='relu')(x)\n", + " x = tf.keras.layers.MaxPooling2D((2, 2))(x)\n", + " x = tf.keras.layers.Conv2D(64, (3, 3), activation='relu')(x)\n", + " x = tf.keras.layers.MaxPooling2D((2, 2))(x)\n", + " x = tf.keras.layers.Flatten()(x)\n", + " x = tf.keras.layers.Dense(units, activation='relu')(x)\n", + " outputs = tf.keras.layers.Dense(1, activation='sigmoid')(x)\n", + " model = tf.keras.Model(inputs, outputs)\n", + " model.compile(\n", + " loss='binary_crossentropy',\n", + " optimizer=tf.keras.optimizers.SGD(\n", + " learning_rate=learning_rate, momentum=momentum),\n", + " metrics=['accuracy'])\n", + " return model\n", + "\n", + "\n", + "def main():\n", + " args = get_args()\n", + "\n", + " # Create Strategy\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "\n", + " # Scale batch size\n", + " GLOBAL_BATCH_SIZE = 64 * strategy.num_replicas_in_sync \n", + " train_data, validation_data = create_dataset(GLOBAL_BATCH_SIZE)\n", + "\n", + " # Wrap model variables within scope\n", + " with strategy.scope():\n", + " model = create_model(args.units, args.learning_rate, args.momentum)\n", + "\n", + " # Train model\n", + " history = model.fit(\n", + " train_data, epochs=args.epochs, validation_data=validation_data)\n", + "\n", + " # Define Metric\n", + " hp_metric = history.history['val_accuracy'][-1]\n", + "\n", + " hpt = hypertune.HyperTune()\n", + " hpt.report_hyperparameter_tuning_metric(\n", + " hyperparameter_metric_tag='accuracy',\n", + " metric_value=hp_metric,\n", + " global_step=args.epochs)\n", + "\n", + "\n", + "if __name__ == '__main__':\n", + " main()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### Build the Container\n", + "\n", + "In the next cells, you build the container and push it to Google Container Registry." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Set the IMAGE_URI\n", + "IMAGE_URI=f\"gcr.io/{PROJECT_ID}/horse-human:hypertune\"" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Build the docker image\n", + "! docker build -f Dockerfile -t $IMAGE_URI ./" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Push it to Google Container Registry:\n", + "! docker push $IMAGE_URI" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### Create and run hyperparameter tuning job on Vertex AI\n", + "\n", + "Once your container is pushed to Google Container Registry, you use the Vertex SDK to create and run the hyperparameter tuning job.\n", + "\n", + "You define the following specifications:\n", + "* `worker_pool_specs`: Dictionary specifying the machine type and Docker image. This example defines a single node cluster with one `n1-standard-4` machine with two `NVIDIA_TESLA_T4` GPUs.\n", + "* `parameter_spec`: Dictionary specifying the parameters to optimize. The dictionary key is the string assigned to the command line argument for each hyperparameter in your training application code, and the dictionary value is the parameter specification. The parameter specification includes the type, min/max values, and scale for the hyperparameter.\n", + "* `metric_spec`: Dictionary specifying the metric to optimize. The dictionary key is the `hyperparameter_metric_tag` that you set in your training application code, and the value is the optimization goal." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "worker_pool_specs = [{\n", + " 'machine_spec': {\n", + " 'machine_type': 'n1-standard-4',\n", + " 'accelerator_type': 'NVIDIA_TESLA_T4',\n", + " 'accelerator_count': 2\n", + " },\n", + " 'replica_count': 1,\n", + " 'container_spec': {\n", + " 'image_uri': IMAGE_URI\n", + " }\n", + "}]\n", + "\n", + "metric_spec = {'accuracy': 'maximize'}\n", + "\n", + "parameter_spec = {\n", + " 'learning_rate': hpt.DoubleParameterSpec(min=0.001, max=1, scale='log'),\n", + " 'momentum': hpt.DoubleParameterSpec(min=0, max=1, scale='linear'),\n", + " 'units': hpt.DiscreteParameterSpec(values=[64, 128, 512], scale=None)\n", + "}" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Create a `CustomJob`." + ] + }, + { + "cell_type": "code", + "execution_count": 2, + "metadata": {}, + "outputs": [], + "source": [ + "# Create a CustomJob\n", + "\n", + "JOB_NAME = 'horses-humans-hyperparam-job' + TIMESTAMP\n", + "\n", + "my_custom_job = aiplatform.CustomJob(display_name=JOB_NAME,\n", + " worker_pool_specs=worker_pool_specs,\n", + " staging_bucket=BUCKET_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Then, create and run a `HyperparameterTuningJob`.\n", + "\n", + "There are a few arguments to note:\n", + "\n", + "* `max_trial_count`: Sets an upper bound on the number of trials the service will run. The recommended practice is to start with a smaller number of trials and get a sense of how impactful your chosen hyperparameters are before scaling up.\n", + "\n", + "* `parallel_trial_count`: If you use parallel trials, the service provisions multiple training processing clusters. The worker pool spec that you specify when creating the job is used for each individual training cluster. Increasing the number of parallel trials reduces the amount of time the hyperparameter tuning job takes to run; however, it can reduce the effectiveness of the job overall. This is because the default tuning strategy uses results of previous trials to inform the assignment of values in subsequent trials.\n", + " \n", + "* `search_algorithm`: The available search algorithms are grid, random, or default (None). The default option applies Bayesian optimization to search the space of possible hyperparameter values and is the recommended algorithm." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# Create and run HyperparameterTuningJob\n", + "\n", + "hp_job = aiplatform.HyperparameterTuningJob(\n", + " display_name=JOB_NAME,\n", + " custom_job=my_custom_job,\n", + " metric_spec=metric_spec,\n", + " parameter_spec=parameter_spec,\n", + " max_trial_count=15,\n", + " parallel_trial_count=3,\n", + " search_algorithm=None)\n", + "\n", + "hp_job.run()" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "Click on the generated link to see your run in the Cloud Console. When the job completes, you will see the results of the tuning trials." + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "![console_ui_results](tuning_results.png)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "TpV-iwP9qw9c" + }, + "source": [ + "## Cleaning up\n", + "\n", + "To clean up all Google Cloud resources used in this project, you can [delete the Google Cloud\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "sx_vKniMq9ZX" + }, + "outputs": [], + "source": [ + "# Delete Cloud Storage objects that were created\n", + "! gsutil -m rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "collapsed_sections": [], + "name": "notebook_template.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.6.8" + } + }, + "nbformat": 4, + "nbformat_minor": 1 +} diff --git a/notebooks/community/hyperparameter_tuning/tuning_results.png b/notebooks/community/hyperparameter_tuning/tuning_results.png new file mode 100644 index 0000000000000000000000000000000000000000..cf0ca1bff0f32e42a59313f5aa74650cc44e698c GIT binary patch literal 272299 zcmbTd2UHVZyFLmk77(z3^r8Yv7o=A~nn)882vwT&BE1(CP>?PlH8d#+q4%nQbO-+t^Su2^z%6NuA=BR6d9Hi?LC+Stg;YpTf0vuL_Fy|uBoBqCD&oRD-)FJ9$t5SB|t zS=l=Hy7G;t8^iFgxAd_mmAN2MKN)DdEEXLo)P}o$LErn{KoRF4MJ&V zc3K#Tg4T%$$w6a_7I?v&2yCeckL|vj?$$qqa%r19{j(sv- z=o1lXW!0p-X3^SV=)%4J#Mu1rzq!QYh@yJOncQ=}MKqO|$)*rr&R~KHY_{ z5J*~cAA|f$wIz!M0&A&@KrCri)ZYMrU)$>y$st{|H?X_ME6l->W)nmWCM$Uc>Y=F`PV=#y70*n=%t} z*s5s`d&|B+USA!Oea(w)nEFaIYUSDh# znQ8n$bcG4Cy)AR9cE9On&}EOEBPT*$g#iV0NS;`exLq^@;BxB z&)d$I(unUr=c=XHYMmJ&Hh%B8!*sxUT~5yWI^$EhzT5QeH{L1v#j-qN`274%%(F+A zyA|6%F(e1~$@j!ib%|S$y^cH$XOL58S&bE565hLoelg1e)x7yV8&~keHiAOQUyGV9 zl09p2$lgsr|KosCQYMu(RXsgIUL*7V5XM>pcW3h_QQqgL{A7K!`@eqcF(gJ=n!dKD z28Sn@gY)cN2<)0Ta`PLyO_(BOYyoSG%vI+v_bKsVtuGyj$V$T zc~ZyING>|$GaT^p^yM;QQGFa55Bb=f9*V^A#Y=Sg-!lqlWQ&P=(Vf-zT}RXQN&XY; zV<#aYAs!*ZGbOw5XRFp=>#Y&V{1eUnhoIsa+QN~gBb%%8Wy2!VxWLe)w zNUc?^QI+wP*D8&9>JuZVl_E-n4B)Tf>X`5`E{F2!NMTjq_a3nxbw|XTy^}`b7K{dr z)fBo`iR#}6SkqI*rRv>`L}o;1{APaDH@G1&(&G=&6YC+E-NO}(14^P0`z>k+v1TtG zEgjt)8IUrYr7>DFvbwgqmW19OavIte9u(Ypru_7qt)`9hXl8owQ+rwlMJDqn<}!;{ z7e{W5uokhtXN?FQrKl5F`Y^BAc1u3kv~b?I4HF9e@=bAGu}-l!ay@d3ez7B}BPTMS z#_5@~!?ci2p!E;U*}>r~eAw*M=V}aUcV6YC+6&cL44d1&rL*CDYGmg#T-p1@l-exO zyt-%O8zf(BMzYJ$qPgd^_Jk~tIR00kIhbGLX#>wI?riQQY6EV4ZXEX}cRWu74_1G+ zCR^`B-x7i-x2;~U7Or8klxnN*z(tWybc`!nS9Git^wMYSt6mUa6W?X?L3@h7mA}Pw!4%Zy zD6X9j83-8=8lW_y;eE>ci}#SXB-DUXgSpx7cw*kfsL62Ou%>R-eZ-X>4c-`AJL;Gj zN?$cb2lRhfxjp}MD(B~7RepT_57(Sk9=`hV$0a1j{l+GZ{*B*ft7g-s$E0bcG5(IY zb$kb&_eg40X@~4Eb+5dd=C-u+iKd?VatYFsDm`zrs9z_wunBy2+b6nGA?dKq+ec;h64{aK1YFaSe z*q!4084S(5r_VoHzVLXV_QEq)cj=%j@3EYY+vr|FDuXsfw01w52`%Wd(xp8#y4Jqt zJ``=nV4nLlGG|chp%Q|@oo|dC%wWKLZ9()M*Wy;^)Gv)F8Oan;@c_5wwfq~7Hw5nt zun*tmVBdP_2fde4`3&t??q4ohC#E1_Y53B;Cv3EIl&WykJTy94Y{e|)UCLtwYx>uW zxMry)EBoW$D07Jp??^6MPxHo^M>bsRg)P!Ta{`iG3^2zRzhDuDq6xThnt9MiILAT2hn5OWrl3PsR3`&+naRt9=H4 zu3TDqA4HN$y+EcwW3U__oU@OF4+#^J+b<$@z?zT4}9Ii>KPkERX@>(Z>IK3 z?W@L(j=ew261r=}kM(!GmJ1Tx;osmdb(;*N-W{H}AotYl9@{P1O>&uW(v9ht7*!wb zi_h0%60#C}$5V?;D$m-ko`rJJ%L|V=3e4zKqetEjtr!?Qs2K%KfD@e99C@a3ZU#No z6)neSubXDPd?$|hB-$k=J=r%TL7Un}O}!T5!*rF3#v^sU(_3RwroMEx>2@_6mK!`% z*N(ceO~*ouw>OQhwj|n;4B)1`V2vcOTb-ZaU>GgYg zC+Z=!5@$3WgOSe{7yGlavqgnY-;_scMjYiDl8za_{jf{l;>Xn^rBN>8xPv;u+g9os zmg!OfygSpe`)8|^MmtTKfkSw+8MYGDYk^oC;Z206>Wn3{mmCp*h;(>38D(^%*=51@ zH%f3^bez{Xx@p$q;YkM~b+Zw1-s@3%|Sq(V39DfZID0*5} zGt)p=Yby;)Vq%l7$A7~fFW;SrM>6h1hcgmo1Ovf@X~IeDz^8$+`{pKYjb9r#Fq|ja zSma^Jh6&+Fx634{zSGUKF|m0ou;LgoG{v|rIVXf|$4YuZ&zra9ZA!GZ2cnyqkpV9W z2SrXjR>P~|d=w^hEsB9nbV>1go_YWm1_HSRHPF9zLx6nX|LeYJ?M@KE79Q&jlFHGrFSjdPitP~)8N$H z`Pe<8*AIzIatQ=~dK|vR{zLcd)l8f6^_#F;4{=fGXi60YP{@3wK zL_}dWL|6WOjXJQs_{0F;i!%SUy&M-xbPf1-2l#qrUiv>*lPYIk{y+OyY=Lt`&$Z=M zRDi9vxr?QxqpP)(+ir+mK5*cMvoge$h=`i);!CXZ`q2(>|KBz`dTx4Zuf)xr9C%GF zoXjkFy&Rk`?n5N$B@XO5Sh|_AcsbZRx{7;AJ^0rZ;=umJVZH|}|GLD@PU?Z4nkI|9 zlZz#bFfTtZ{{v}K78Vvs7mK&zuU{zsyEyPq>VdVJo3l6{pQooMucsidlZzGKQ!z0y zK7Ii{0RbN13LaN)M>kV19!J-Q|8(u|dD9F{)Mc&B)sMJmRe;VxHh5z@-e;1VG zyLk8iWsCnZ^uLY*mX;=!;uuI>nuTk@l&ZAOmX?@%}$$fEV<-M_TxHyWTU zqeNoRQoA#|*ww!*4OD!J!;}Da_+4tc+nyoi%l`k<4vQU;4_PLdkPP@Qe`!V`67kcc zDW@xs?DqIC|E1+7bBKt$kG#fyTm9vWcTushB=@$x`paV0(U-0~np%XILvR1p(2-@) z&6KxLQ@u$q%xzq1h0Lhq;=lWs?|RD`b5S{t*6jT6#;t5-C19mti=w~r8W$bh!6+!m z!azoTgZVE9*;i^77D&n$s=s|s1lg?^dyLg5g8bk3C0U~2VZbt~@nx6(@;NKri7&-D z5N?(Q{-yuX)CVkMBLM!pPCyh~%_}Qgj)1ZrQT}D)zXUA9t=2*Mm(MvkMsi)b`cSa_ zjP>stU)Bb&%=d0R;=g>($@(9UsABwb&mUt~uF2H4Rc{ZCA<6Jzf9Vr=Zhf=36i2F* z5wO_#6eZr<|4O<1Pqx@bY1?$NRmQ|p)qwa`Ru~hf-iL`zg*VBTiSK~vz@@RlD(zjU z#KEBaRBv$w^sZizo#mwu6=GLNuXKE&UCc{G-Qle9Co=qG@Ep2U1!W6P@fyk=3om*tG86yTq!@eQs(N>3U2AEF#j zmXkbJGkl9vO+#t*55MV4u$=58Wg0LgEwfE0E2&#e*jd*PSIJ!j2SjsH1J6!oE|GEn zsh$h)t?7CY@m0J-*Tf$sbvUZM6)7_7T+(t@Kk}5j`FLTs#WKlg4AFEn`5lMf4m`(f zHEcI~onNJ3SW}E*K(C}a0srn%&v>t<&wSNZhro~rtpq&QQ)Jrt!<1y;s5MK0jr)_3!D93J~#Hyou$`J!LeD^CQ zLKv6!g-)o)$Ya8CYqMr9sw5dHb1o^(LB>-VV_+IWwSalM*UeEvh?l#%{2qISBjt2}edAN?^n(h3aWIVZ9-pl)amLqbjB*Xj- z?LT;PIyU~HYmX{gV@KRmrqcu8@QuTL^A*L^-3Yg0($>^%Tx37a-XYDncikCrkLamy z95s!!pBICEc=lrt&03q984o*9{Mb!p>AkLpjsTW=#~g5M+HbDerPk>+tin+Td7J!w z#U-EvVwJvp53KD~1lGR3yVuL-sjVU9j!;-Ylae!eFU7w5#9bmoB8C)OWf-qvLP^PH;ZGI_Q%)@e zfuv;D0kIh?sDGkZ$?RwEakGcW&jW1JxY0)sP%w z^qhW2g_qv%6YNB-=y*GJ#jsMbD$UUANNJe16dkpk6Zj}dIh|>QLF1Uw!gycojAwe{ zu0!GXL>Ee#(^AZ26&$LwTM+(HmA5VzK+`V-jHbZL%AiT@gvRY=Ja2M72X@xCWa`Za za?ZT(CIx1BaxtU|N0{-F1)3`RCo$klewVdeFrDH9WC2of8y_7+QVd&-KxBqpXuX6+M_G&p@{|hA=RM>;mIbWu<%JWEINCaHsk9 zYHrm{_sBH{u4ikhp^YZaKS>ttloIILykX5S+af8SgPjOnN!y8*(`CLgRZ;l_CMw69 zCD?BE)V?RvK7-G%g+zcQb@hA+hvS~GH|wc^(8q&C*s|i0Qkw~&ABm)Wm2`3oROys_ z{-3APH*#dZqfOq?i@RCp2HsWmJwAyY;`D7a-gRoaJQqHDew_b#`If<&GFRdalj(Pz zj8wf_^_b#V5}tU4faCeFE|{bub%APP#d=Kl>A_txf3{u-3r~5bo<#zdle%F!(cl4D zFk7!XdYG*|2=YtdEQEyfA^amP#Io7K`#hb)uQDjt#LC7 z9R-Ons*DObjb67br-7V;e&5qdGAq%vVueGz=k-}*FUhJ1TWTI#P-O78sV zdf!GBF~N_e^4IT@h|>kzxdlR|-q8I6%DF4|<0SIcD`J(_*s~(x^VlGZ{mn^S%6#;qnPa2vmZJ}?FxPWQ=on*8564Y%La6m0-vB{`8p40*`ryGfyb$nwEk6+rZ^Jqa z&fa_BW`*#LK=Z{f-*fC+N5VAOdmo=39*k&&rNX?5cfNrpRTE377IZ%KU+X^2x#Xn8 z@b1NEnR%y$aTBph$e_6AR$7r&8=LM%Qeal5Ax)h9D0Aj-=Y{U-lKOSdmb&mx#f-b| zMgCY1VMEvcAm{Tl0@pyd!hIv1GLtif9OUQin%T|=@qX_al>)LS4f=O&Wg&dR$NjFZoQ>pmp(gQgeN+< z1%l|`{G;d6e3MOf;ko)y^0MVmqa4ca1D2#MY>>LbD(}l*t2md>KMF z1e8$^9=C)OG%MZ&RlEdu`%rf{J2juYZFGv@sn!#aBD9CfN7j`14<0$JbspntVw8P` zTE~yfy{41M`Dn3K#|se>^x$&mrXmucGWz_0gN#b%{zXDLi`-ate@b*L z#v{jK!=&YT`lz8MGurDmxM@;kI(alBy|s~%22QfV)O*qQD|wZn7xG`g005)(PWr2g zP$>nqRgLku6p~+QMh&J0c8IDp@-ilk@}BQoquLteR=QFq4ppc$kVa(w zcGbA?R-7ttq25aSU5!@*O`!Mg)HuX9Xn<(~WCkEsm^H?H?RWnCq>6Y}-x5CH9sHkT zu^%6)OPNO($7aKR=egv{#7BDf!B}jFq1N9Mz~SQV>tA)!%P>Mr;_m>A70vxod%R;) z>9;jf3j3v%Q*B-8eB;_$S3qe$W4$JQ*^js7l=@1WlXfMM<5>TGY5d0S!9k!7v0X*K zuqr3-$?{=AmJpBLZHO07lm3bwxxW`+1f@yv@8}GJKRo4w@)pnY``kzF3SC)zt;z#J zMdQkeG~IrZ+#j9Mt>4~p2ZMfG_^8+r3a-$fqaGlm-d1)YxG6+A(=3q3@UA0l{R9YU z(z#`}8~4HeEr5THKM6eBhc2E_dqi?u+=ayOC$FmMKwh=U;@xHOe^a^;(Z@so4 z{%SFk&B5a@6NXhIOaDcMc1zcLBPf=^Q!_0v5xFD%)->6xJ~RjA&%(e7>ddjzmv3AF z(W;79bNHB8dy_-$=UXlC>#C@gXRqKS=vELcBT0&BE;!`DLoy2mAIh*cD!VD1)pOVy z5bYgASV=Q^{R!@tW_Iaf-4y^L8UqW{1=!+{koynfYwg>7a}!A>0kqxXc6eN0DSx6# zr9!k||1<4!sOOBkuA>Px=a=p05bRf4tF)*3=8-}VX14eAam{AL-PWT8T0_(Z_P5n! z2nXHGT@b7!8At)whd^<^x=%@tn9n+tM&v`j2~G?MjT>rLg>zoPxgzWe3T6WEo85ay zVRxt^FkXjeLuY6c@N#1~)ka7NN&Ff2*Hj=<6+d-+?L^YOlTnOy*O~&D^mzZ#s@ds0 zhg9HE(XtBOw+gtp3QZN14R7rtv=3Mg@BYJG@Y?!c5 z_n(Q7;2Tde!{y)Y%fL>?H_@@q7K=zQ&6hILxY$c>;403Or>OG+I%$1SxiWE1V;+%; zmdv^~@g4YA?Fu-*UhYn2sD#cM$27EvxeSQNd%9qLAnyCci%?9b8G$8Z{N63id8OlMg;eoD7A%0-!ymCUK~H9aFW!MnFAk+4zm^4dmvUG&|M z(ux5PA}ALGir}qZ%YL+Rk4gU6aJ&JO&?l7!upqpGzv+*1R?EBThJSk+A9+<~Y~ASP zYw`PaqV6=_GV9o|d71}fY75GJpJAHWR{{+<#m}8HVfW|5m=)`F^ML2Z%#II(on&9L z&hfuUn!SIn&niORWB~uZ%_SbFK(J*N`snsBrC6$^#78GnWac3KC!aa$_DBWX7*jFJ z=ai5HUr9>O#QehM`#MgHrSTgdi-9ygX%m~ZMAHS-*KGpWK^?a{9tbu85E3aKLrHS_ z)LL|haP{JN= zYhw5<$TdH`8iD2ToW73u_Qnu>mlj00K>YCt=wnG+k(uA6BI_!T&8=Shp3U4M{;mpaHs1@sc&iB4-@8`4O|B$IQ{ zq9;dP86!Be3iX#Jte}h&nqNUwY*a7cC^sg8H)@d+UUUadhZ6DxqT9|@v6~ZcB>VHL1TIhkFDXFo2rJJNbaqb7nvuSyoFjy8Y)8y0s1#fqUJ{?+Q+k`8^jt6Za z71Vwjv@~9-l_b5+zV&(tcSSY_T=`I`RY%%u4p|6skW-%}ZwbO;pgZY~V zY0Q@tIky3n7kKw_U8o)iHSDI%Qg*>!MgJgu0~mx7{fjg7_Ypq7#~zk;1h^L+o{}4| zn8hT}_n$>?!Z@c+egr3?O20$Y+URl3&kUD-1Yd?zYn#k#!mQHW!6*TkA5Ex$1_0q; zmd~D32@sAB3Vs;78qn}-n~-i=|B+EcEbLe{jlot&-D5NusZGYuC;0yzs~yGYk;>}xLqMk-a>BLvxIrx@53a~Ob}^6cmGRS(DQ@AuN++}FKF zneUvq=9q3(q-@0o&=Q`$($Z#gX=!At*eX%*w{C8(K~#7_Lv}c;=$koi=HR>uZ{${UaX2_&wCMeElBQf)vab&G|RbE!Ac5S_^)Jze?_x) z34N)VOKfEz9T^3hlhet1QXhjyoBnu(FJXZo@?J&+oV%$HBdRXP!H6L6NCvU;2SoV< zzn!1g?0A@@@EaA~W!4k7A1dnb_yWl7QEyRcD=BEd?FOtI?NI(+3C*Gns0CcY6u~If zITJ8{O_K?i>5f)sGF4B)HftC|oTx?mhaDWqa)____<B5Bs@&u4PFMOfuyV}pj~Zi6kp4q_he%mH5s&~vfNxO zg|^-S20ncf6ePFCjXY;8PN;MZ%WUrtW2wkX(y8i?xM1&d=9=Y_APF0%T@tZ#Q!6-${WfT;5!*OSHd*sgW-{ZkLJi_&pR(d+UUkdOegp=25f52Vc!f!w`v!o zEEq#ldWA`DA0{u`VuF=*}HwSbkT!;BFt6R@`ER0BHNBRmCm8NYn^%`2NHT zGXo#sR=g+8;U{xeHbgiH@P6^1N!ZLPl1WQh0WOBR9;Ondy9*|E(3tAsK$@P%HvO>y;2#Pr zkd$uB=r)vT1k|H1$WqwjqM%0npe4|?Ur2XrffVskeqcT9%uV~d3^`kIAZfso5D2$F z1$J+BQ@xfvyd80{C!XKj0_0`Hr6ib`CXi0i$0i?Gd5i^ozZ~ZsSTsLQ30$%G zxm?ZEDUcTkgUVV_2K#Aic+mE2Q#@f8PbIwtqy0J(sn)`aWRx10MLlclWX5wk8N$*6^YPOl*tSGnNcLG z_luO4@7cX-Iiaev?_7roH@n%9lnqg55UeOLHJ?_F?j>y%{B{fTP~^r$y~~F_nRQ7E%;UlC@?gdza$S*bM@9&ZNu{ai30LYJ1Km zKOJ1AH-QfagW0s_pqx614tqD?RJLnJjbkaMjrvnkfFp@BAGJEbl!%<;^(XjeKT;@j z-lPjytv{NyvnzWvVt`3o6SM@Vk&@%Z%YlfHXxv5*7xH%!|jE&pA!p%yz$7QHVcB&AKvN`Y508C+7s~Uo&pD~%g z3i;1&gedsx4L41hBlFaXm;|OY^62|A)u?mH;lQsFgpTO*(ivY|IonyOgGLuQv&V1K z0a$N)Dv%bbVA0rGxeu|pZ3UuTwB$}(D6&;AZ^UT@+SWIRQZUlV@Zpd^)@z-+=nkI; zfXEwwOs+I!8Hmh0s1H7w7|^srtRE%oVm_0PZ`dEtT2zzRrjZji$Bca)tUv1m84j^jvoZ||x8!zS7oK)*^GXS!NDryKAP5Wi4h!FA z=~_`G4@c-Ujy2mtQTDYjtz?dL0dBCg$8A6^%ke@ZCPSY%1-bsfc;f@5v@?)2wK|ow zGwB3W1>8%iZZ^a!R`bnE?zqpOd#==ZksMR}2V3xx=9hd6`aTFfy}^lFsUSk=UG9y= zM3aCMuh$PtsN6-Z2U)CsW>~mVo4KfR@8YFUt#1Ze4R`&F&V?nN%8LSDzlBytRhKQr zsp3%SA+OUw9FZm>ID+9(^F$i{+bpvWv;ab)XO*C|xLS2)ZYpa(xs&8hO-O7Q(rvf!I9J_>ngPTMXh-SMuG*@je*bFaJG7{* z190pQ*yA0%EUdE5kE_VbHic)s7A-QiD(a2>E7scw-m9Nk_AB|Tues_t54QU~eAkmRd9`Xxl!N-r6-e}j zSFDHb<{gT80r0B!d!l|1q)VoG4#Q+Me{L~e&bx0H zl@jg4hxU!wYVa9tLcX8u#zVNy2-|_402NnPK{_Rb-xZ{rfo>-${R4}jZzM~sB1DGx zb)I5ewRwIP3wktF{c7>3*nXLxte*|#FpO$=mYZ+XwcC;a)-T*j*Fo{O%xWnyg363t zIZFyZ0AQ^p+bm~$PA)2z#9+Zz999H|+tg~dsJk+an* z92qn=k1;!bqdv2BuB))-9{{~MJcv#*Cw;2)*3tce)hRIxhNNoJ# zE6NH7({c-A1kBp6XPr zws|kzY}Ll@vHy_fA8N@G-zlK`MwtkCa~nJ#zBJx_#Lbd?5P#VIe!9EdWyrrz1mxda zQC zko7ZK@KjNiqv$~kg%KX+FzaLz z?E=h>x2EVWHNC1J@-1%yq#1R^AcMuUk8?vhw}t(fLnJI&y9)qiMz{2MeN7l97n&r+ zfj%55Nj3kr_0Z2B$c%Gz=3}(5KYXTkJDH(H^u|i_uKtcN@djv~ZG682OtP>imYQ81 z*8srolqsXpwX>9_heeHVGtX#hR+(}_O@HF1!V<+|XujCy|B##hu z>U$6EhX&K;dA9**-R(m+#%JGugZktU3Ix5@K-2|0e)s!)tw@P*t3>wg*iHy3Dx^2j zznFR1P6)n71wr+t5Rgwh8Dc>lFnyyeHTVPFeG{MKNde*tm8UF)rtVZ;H<5@?-?7qh z(lj)~x`}8z`Xi|_E`Dy>PC-6*)rc$Q{?$x6Z^W^G4Jqt10<^mypM#Xfb9-xW_6hnN ztXUWz=K|nCswujSr3+Ay#JY{IVO=NiZ(X;{JDGg#7m}z%YivWcF7TP98zbhJYDv4G zgERiuxD$;ckkP!ZZ2MauuBl(HmE7;U`($ZxPR8_4ArHx;!`U9+=*L|Q)r|n( zS)<1N?5kr_KWvZ<^W-Z{^~#k;BgE2h$@?D3Cx+YfaOZCuVNoM6<+r@488o3Fsvcfj zFhp=7s`x&mxl}(4qX7Oc0vYF=M^)(03(Qv)#@K0%-0GTiI@iFKXXnLaniyj%eWqNI zQb|Sm=DM|&8Dnd%*+`X^bKJu(*>-I(MfP3;w%z)+*0bL|8FyQZ$IM;HD{#Y=JA!K1 zNr0@&8p{^SpYxb<&^nfyoLY~5R`kI4wDWe>JOER*DoT8;^hg$#()8Gd>)$td`van$ zMu(^WpiT@fzlf0wFS9*@9J!6R_)TPL9Je{vzqg=S)8a3i&3kEdrnCU))8J-RD_0^o z#Y^v#zMmMJiM^DDd-vO^S7vf>A+}Y~RMknhQ-1{tQv~p*jgAmAC||p@YdboylJ%2d zTPl!KIkCL-HeO%$_4A}1{nB}hoK(m9Cq1P1CT^v7*Og@{NIC8433usekaXwn?`lkr zR4rg`-}m-DMtn|0?^DzZs3P!GA0{in@?2U-)0Vz{-OcxkGXVO*AHEiou< ze@Y{)wrZLg^ju)N(^2!9LAVf62MfsAzlZb&Do7k(p<7B94Ye7-}9J z;V=1{oG-Q`FXUL+YA;NGJG7CDa2w9@Wm*zE0MQDiOx;e|(CQNN`LR+tGa3eiNaSFI zpOJo8hxF#>CtSFSw()+UoA@2%p%NF20)zYv`zlg{KFh?Hw#JbTuG3Mbp8e=@QX1D# zO;x1WA)tpsUpi?e0U3P-iMF19&`0K8P<5|6<&*{773mMk?fKdHSZKjEBRk&@g+UVg zGP8MMvG(ev1=bCuwl)QrozAU;(ti2Sn<_^tc(Bu9Ube${N(n1ONb-9qGwx zSz}V}*Yv!T`0~|9C}&V1KVxR^^x*MlAkl!kOcB^fDmXsUy6r~CTmQ;#K&*Scg)J0) zb}uNy7FKxY8cDkW{K-c+XXC6Pm88Msy{?;Z%Q-;X$+KQQM5wa1())+i2Ekut8z`06 zy{@a|4cJ~bP9( zGl1H*E}JtUlyP&kP+Q=gQGZXAN_J%gzq6@Hqn_=JW;=y5=#c zHu*`iqe52O!5)xKcLj{tmtqVq5i&%~HNS5_Q&YUl5Rx%$wl?X#nw+8y3?b(8Muag} z!1;mKe!mFE>hm^x3}Nd7JE?tFF__Dh0E^~-RuJxLYIlII+wiXt!#)k^BMA8oDclbU zySoXn8r_XRV^oRX;@3}teB5zHg)$jQn;Y~E$+X4B8AjFtTbAJ+$!RGt| zRt-ap+p5E|RvXGv}#vNKDz1tqNxb#C93L28K3Uzt-qd0rQa0;;&5lHwjUqI)O7c_=(&dnKa!fE=KzeI-r!FTN4)uTK5sybEDia|bT~Dg z7aj;JSi)j})BT&A4D29Q0YdFe8kqe}B89B2@~RtI7%?C0Szb@lY!!m!5fBJzYe`jF zD^HXP*xFH+0V9d)Z1My^BwU=jjPKfSce`i3wDLyu9l`_dQSXkY$-P346<{lHJic?> zzRcPzQ_K-e%)T3C;cYkM^8q72|c?Reql5O)+P-V)^S8mblP*neYj6G+>42Rx}v9@p{K zPY`R0<-AQT1Y$m9bwTJN#Z95WxP{$Z$C{u5BPM^BnR68*E@a-;3E?L|vr;ksj!2}# zok;_eMu9<%WB2a)rc+?Vy=o^!-e$;L*2qtk2hdI=;LWQ8USMTTF9o zTqBeaiX1T&1jWC)-~)SSMcVM+VXYhz zfe6e~yRI>TPS^R6`|JcXQ;Dn&K+L)TiAU8$RCR?Mrk)u`fajYz9-nF6k|=39Kf|lc zO@?l%^*dbX8HSTW!_{l@p#kE_B@Ru1&g@_N8paYt3kYld8d-_I_l=5{4)u>PvtI1Ui^C33@&aR*ke zu2_*KW7k%%`aE_3WLTUZ&}zYr0F`gI3k;^GF=7P|lzBXAR4s%{`v}z~b$m_eqKMGfWS?px!!R!K^oW=NFSs>zUf&4P?$p7?)uH&)v(c}dwE+^;z>XrD-myJ z5lZ{^v0vav)^V|wZ#l4DMXYWWG5bHP5#%Q(VTT3fPG46MURi%~_sbA<1hC-3{o}kS zxuvAr+8`{Q5fHyuw%<{GB(n#5y#VC+V2NZO6zKNq3xIf92CXf_!sQo#_^=|+0Z{XF zm(#_G!3oQJfhvi~Kz20e;MSjjbV})CmSN|L$-IC+8oFxYz*>g*fP2LSsOW&NCKj;@ zah6rl{SE;4{6%n}Ycnq+C^vNPP^JgtXGvax2xy&$SNR#-ROa0H6Rkk}!3`u8;&zc! z3lFysI02&3s#XC;3tJmm7qeQhTxKV8pk&!ms8PM!=i#V%aUUk*7JaAX1l<9KID=Wz z^^@j9={_GM9I4p!?>=^i;5Oz|`i>Fjg#hM5e?Kw-8=(ms#P~@s2#V& zcIjyc=#o_4L zJK(3FA5t@bkE#zee)yjk6Y;Z>h+CT;`0MJW+eX zFDj>iv6de4b-|mu%UO7-pn^F$J>+sas5HFB+_c`S8AR(sz4PwK+y`WGq57||wcT|i zqZcKM)ZvNGaPDFa6NT#}zN-H5bqc_7=^!T8qYsr8FJlHm>pZ|lJdhOtTe^bLyxyBo zJt|(Sw)bHB@m6=o3+@JN)CCZUSU@b>2>}UP&F6Uu!3}A#PM~@}k%0h1Pw>+|NJ6GX z3OVb#g=50cG!oqfLV0OYMX6Tf7xGnr6}95nUwP(jVAUoz} zeX*)hPLnS-3VR69FwVk;_P@l>LPGu@_TDqBscmZ;78MH!h*AWk+36@%DM8s3rHK%F z5do>8_mW@%lrGYwib@Mz2%#oQ6%Zm~XaOQM(h_=r5b`bdInQ&>yU%;Jd;j^~KmJ>g zTq|?UHO83tDEAm>;S$$RZk*XJS*ncW?Y3gReV=}dn#;v;z-uq`J11bf*bD?ixLFl>keyTd*1F~doi6Aq-AE@LS@7P6| zB!e6t$vLP#=^%NIX}Mm3I3#cGp6eBkY9pUvG`!Tm{@Y*-eX_9j3<97gwULE4Y(K6? z`R+#J+#1B{_I9g*oKN|wdUqw@K!y4ikZYNcOC`QDJmjj#PTl+5kY%*;;FcU+l;^&2c7DrVz0D~rgd*k3jYwkmj^fuQQP=wNA62ceqt#ywLr&C_u{a|W z8+r%2_NX>AK=27L--8xIH%trc#TGe9IlV<_j1EBeT9J9PAOn;yJ-~;ObuKM%4-&Lq z<-0fmg}mi1?3U#m5z(m#fmc9hDBscoN*wX#4SwhP_yxs9>zpt~ZxI$nGBv$eA$fihn<0<|O5MP#RXu*3>ga^mBXng#RHp1rnOG_;I zEkHSwL0!IZdLiKJsP|-p+tXb7tT7HT5+25x;6l^jKJ^7i4-xRUK&P z_kZrw?~q3=(WZG72`yAAj)@o?P1ds)tT;r_!XY99Q>vDpLmtp z9(0ryMK}0t0l-Ijn`qbFYxkxg@!*Xn11r(>HqWs@Mg`9T=#Cmi~X?p22ou1(86QvNf{HN zpPSYCfRu^VRc-ejll75iqZP?=1oNj`p=4WEtEXPEcQ3|GNDjaS2)hK0MMH@9Kh9cj#6a4%3 zex%a?dI5v~ZLkTrbA()g83F2bvQDE}<|xnbm=X=j+r&|ZXDCcbM@r^Pb*^53*Vh^N z=>`fYOKLHp*fs$yk!zFze`O{nH$hjlf$x{uh-Gif?oud9|vSc6r>sfWtI$D>Mi>0tHaIp=fRM^;#y%8R#@#sMDGh?a6Q*J5~ zJstro9(Xq9iYR9p)f5|$8m1-8^eu3wd@y&{2@Rnb1b~TKbH(Yu?|LGzVejJFghTBk ztV`1$zRR_e`Ig@tWpnIiZhl4H6YW4cfzpBrY*XxGm%vh>&SGz<g@?`CcVxH)O;4*gb$0bQ`?4TsfGHTddxyg%YKWU4I2nctg}1H;gxWCHekNcjtB- zr!KWX=1}5DQrRGo2nD_GdIbLHO;3{@wtJc0a)4q9(F3ZSs=yaw zATWHTWdHz6%7q?FfSmD}Vmm&;`815$8aQqfeM$>RUnl(8U~WG6FAgza$=;>&nZrN! zno)sqe;%g(@bqNbmB^tQe!7?K9=PLb86j^bO2lH$u+ZQWCL4M!(z?&`AujSQK=I+I z@8Zza@ft8PCP&VJ!hO)ZCB~)j9+N~7O8TAmaQ=jqSoJBEszRGj_HUq~7Wy^CAiF@O zS2yl|2nz_XW$Gx+6~wvpf2re!nySW{JLdNAY5|p7K(&Vv5}=;7r$&Nb`f#L+G#W}D zb}tp2)QCC5%3_ME^LB8Un0|;(MVyWe0ugxc%D2|gH%(E1pe~MI-s5%ubvGiSU?L3< zqfxh@4+Qf9OeT-}^7lz@a5I>Y_?cu%Y}4+N{oLgR`4+eyD7b)ppfZi`BVG^O?KTRO z;tRUrGJLIH^ehhxx(vh5RG^gG>#sy9w8@t-Y>X%$kR0C0%pA&y&>HYR%ed7>qGMto zOEG(k>HPHK{7aiR8)Is!!wci;-JBJ)iD=FW6{o$&J!!=*ck#Yr7CbpDp&Zf<&-~DH z>SO|ChV>`q&jV~}KMo}}cQf4PLs_yQH&>;Q`C_5JB|t2=H{w!!Y5>4rnYwG+ zJGoNQ7-rX34}n0f06zSZ2<4^Nm(eggKX+}lpQN~dU&kYLM(tB{w{yb<0ghWND(1(2 z!1~JNWdL2bu@eoicSmIT23hKz7ch-K)vbNykyHMwOJym=OG1kcKplu~p-vU=PjYGh z=v+{!dOs2HwQmXDmJ!xyKe2lh_?a!$!|ky=gdLzJ*f72k@EynYYc~c_ClgAdpNC%x z`v7o0vm#hGby}WFRD3uHP{n_8-u=T-`*P&)rJ^u9p5=#}z;8dgeNTn`TqFxW5g_cDiafpEe;_0?^@sGzJnJj~8gTlCwVgMdTw!T@W)bR0GwNT}nq3J@Gm6uun? z@U#J$uS~AdM*g)GoCPTDJ^Hr4Z_T572SrLM9xve0u(lfp+E>GafzmTACM|81_m2}e zPn+DwL6E+3V?G90qxgTaVZb^LKcm$BV;#GdK(_{(39M#cyR}apoC})<9KavqP3!lN zSR?b~8&IJ2K&?sb$AhLoA&1=Pmob1#TINmYWg6!m`%8TcRJ8Wn9vCKU{9Yo%K+h)7 zzdQox9xl-1FL-U5DsIUzs3oxsRMXS~em+2D(gRpnU!$ut2CYL@!h=`ZIH#ytKt$gX z2;|{U!*Y;bl z>zJ#0qFIk^xd%6!=Lv@Q=4i0vg+x8pX`~ z-F5ctw_aH*c*gsaQpG=-XCN1N_(FG7rQ+XQXLTSTYLGp)$|rw(f}aGg?iv9*V|q13 z$v;AgQbe%wb zdh6OT&HU#9v47M2E$IMC9;{%yDDgMV3fumPdEvpI+k*f6)X{z4zT>!TVSm?d-USva z$jbN=qxDAtQwcr|9AM+UXDHEsb7#VVg)%-j`QzLB&qw+{)qNymzfqcB`;hhDJo&D$ z0q2Qk>)(700$;`t0}uo$)xP(4AxJK;&|Bv?{?FzAIsDiM%$t`0sc*`$IH27fVyJ(~L;BYy zJh}xeH2gQwKZ@Yb9{86E{x_Zef1`qibswb4%9`oJwC2)=T~Xr3O;D|W#O@ir zjL&ClOl*zh+NS9=;m86j-X-Pj_j8J$<;U3EsLAQQs9@R=(hd=|gl^OhKg{mxaih(?vUQg8 z{TTeMoIL&|XH?`6nk=EEREAjm>}J%!Oj~P5y8U>TGv&+qYLMu}sGB;Bsr7u(u)QMj zGLWZqw9PlWX3iplV0$OH+#MBUELSJK&b?Gtuo$#gHy5-QJ%(K8+Z1b*kZ>jsU0mc_ zFn25`f9IaUJkQ}*hSK_Jqe_@vC>4TuxI{@H=DEYRY$eR^~!db!^*!t!bfa-;FC*kUJl)pNb0l|_m}m(AJ?Xm+_Gk%cs(?F zZzXr{{MaY=jy8!@OS>RB7+F`mk%Q{Bq1Y9`L*YwsYgr7fZ4-Ka>ZFBv!!h$%2!nxQh#*)l|MIf`f-+vm>GHfi}&@l zMf+^iCB#j9ga2aF<CLI2y;>-{E>h}*=ZU@=ASaj~4%Zf&>i)CP=2s}2aV#6P`p zeoxr}owFWDq+4sdd652=pR)bkYt-jj9OcK3LP9?uAP*0sFRVppk9v8zlfO3L>(<=d zM&!aJF$xvNSZAsjo5!?0e_KDhU)@I1IJo;MMe#}aE8=yRzVt_QXs=py$I=a?x!C!D zxzy0*?H9xn7*o8p@Tb5TR>t}X>lj~!Z22$Lav|C@Gp;Ps++ln1e%B{=nD0n7OX*V6 zaHiWDS~fhqnYxVY4q8;gNb4?aRC*W>@dZ02DQwiox7@D74AvxICKT46rCwi)h#ymx zi0QJllA?_^RN?HYZf=5%K2J->A$pLHBJ`Tb_b|jQPX_Fw$7Jwra)Y~Kq`{C#Il_ll zwWB-%&%Iq(Torznq}UXn;oLRyu~I_p^9J(m5knu}??&ErOkyU+-{P7k&y!2&$sq;_ z3lUSCn-Z2yC@4M5O-_CbLLQ6Xh)kyLXi1C?XEotNR=R^A)cQzwd`l_hV~HpWT6?RI zJeT6qJIj_N&M+$^A3RZ-6Hz=!-JN_wh*uq-2*mp4Er=8g6(W1-+%Hz`*%NHV6(pSJ z)_^u`-nC*&>19}1(4-|9`@?!_*kMu1UG0FPA{kdIm4UN=;&xjux9OH+>e=^U=0oE& z(LHOPOwDwcF^3lmL#`=4F~^)(ggn3dgZBP zvx&Cz61=A*zG5E-BXb5Wd#~Vfwz@27-w85@HtWUXD}o0^YvSiP%<>4^gV!veeO!}M z-HhftGt{>S%Q(jee8z-YgI-u_p=4Ig9OoQ9G+{;@cDFEn;Y)m>!#0dtVX5fsE;YMUI)`T-H#p-8qVRpk$>C$YQ^&T8794IG(~fdLvtfqI)B_EmS^8+| zdw_m((wtEpb{8xfG(n#u=VlefbW!6g20 zV~KidzkFIILX5tYWi+U4D_;EltV?=xvICABZ-y^t^H3f|!PYBXq{l`c>j&lh6&K*H zC;*Ly3LiXohn*qFD(qU47ij6iT{HCC)suPtA@gokAEMH4i6OH_E;IAgsH?1AE=Cbl zQ{j@EAL^pP^gI%k9v_$VWS8PM=Vo698H01{`)5U-zRthnSFi|HxWFo5M?0=6)N@=D z&L7UfSF!vqU07+{WAv8WwLH&zV$xYX!Y`-`t>e1b^>58o&C-ZS>P^b=B4f#8HNv;8+$OS05~kVxLyh$7$_v2mjeE8Dmd8*~b4Lmh zHk6uO%#F{FY?^{r3O`2iHfZUQQ}Z(5!aGhAwl~7w1gZ&M6k8jMpBl{%vOa=KPdb=* z?Jfnxh|jLXgn&tu{5l;Pq?qnDJ*-^8FSquqb1ix=@chBVd97TONW6)Fl%bKH%00T) zZqknQaqnV1uw$lk7qeSZ_(-!~LO7yXIDew?J!av!X14nTwc=oYO>VI57V3-im>gyt zZEUf0uRK)5nqp2p;~A#96Xz5IqN54+J>z_(7*M7 z1^guQvyWcy<^Zn3PNs0mJ$8w3@Hgob(@w#I%qQIf*AI}lKjq&^fXC$vJ2?aSL5AUy z`|J&?Bij9XHa20*WdNlCCt0mS_jp0fHHmLN$hg>S`=;axpBn;&kWA;~Ts|IaE_im= z6SK=F@}%ViiV|96JNB%&R@1VA>(P64*R%`5;M$=o7n*1j(060)yQ>+L<^HS0h)9%X zM6=OJLR1TS<h9)|QukeK;hpWCtBo z0AH2L#DAr-P`<8JjKFgW4&Yx=WWhj(g658r15vXMmfaIUrnzV!&GO2-I4X}wj&W!l z{wSMCW}PbD77Fu=rjPB1++qB@J^wziW^QBMY%r;;w7LVUA@)Q1U1NILBUU5(&5s22 zdMFJ7pU$7NQ3E&WB6Gwn@^T;aFC{HRnEE34Cp&Q?chTGji#q(fkCfP*Sk}N3onsIV=o( z)f) z>+r#K!TBf^TSc21M;RZU;8%(cy;MkD%Vs6!YoMRyh=eo{^1WMCK_Vv#^PT$~#wAB% zcgyDZ(BrFBwW>7>e3)*E-tREvTfA_tl{>6ZvUEwcUgt;H^tc0)ABowCA;wvAiFY7l z4DPANNy5o9w&QJ+L152*<6LFuf^4@~8ih)ELQGZ|4ki;;2!l*vN^0SVQfD5W}9dPqz8YlB|f_#0>p8U!P=Kgx}oANY= zpWLU2C&1b8vGYQBM#$ZKAcJ$oBhB?g_SvAmNa73~zT;)wXl10i_bYs4ke#U5fIi&*~ zV-3GGiRchDWc^+@Kh8?Dj+ds_kfx=-P~NRRJvWS0#yGD2sG?ur zjkO+f%``z3x!( z`V~uBLFCmk!VM-;U%z<*8RgBvB&WX?XeBSvNDk+fChP+R6|F;w6MK=E!4Z;<+@^7Z zdbD(H zM$&aPvIE|@?=3psrEYm-caDHxj0cV1QD!N-9+WG;l|hs!5zj)g@Yx$-GG3)OeJbtJotLsGDikOpPKv85h??B zi;*K0cofwRdl?*l@1u0MIG>`wKFQhL6eKbopXk7iwHzA%k#g{GK#V&de zP>5NApmrQ%F*SL(ruT*14cx61t#}qwc>h${+4=0iCZSf}uI#l|_M5(?x(XZ9#4;&L zZJbQ610bx*uUBZ>KdN%ZbT2n?qRm?2M&d5XM&<9jyq#hAiVR%4Fxx|fWY`CEnBASa88wBs@(!e>H|HR6N=;QvkFv1MHnpiH_iy+^?_NY^2divB4QLFr zc^L43-&%9vM3aP~Mo-=E#+Tz@;*RPOFt1U+4BSO#tldn_F4Rakq1f-^%_fWZ;R3fH z{V6mcZ;fPUUl`wdAt>pGZO%)Zj0C#Q+|9uEQ(KtROD+3Yo9jp@79!UK%hYida)?do&*>NVfv+? z!tv#d?$lNw-Z1FKprz=F9(TQ+u!_GdzBv11_<@Tp_0dm>5n0txQfpk)j_I4 z^CEzHZ_|RdqAzG8!$QGl9#l)qdctYZ04buiz&iLz2ce=Hs$keeP+ zkD^xRCpez@=2RLsJz-QtQsX%XhG_6y^65ZTbzr((gB=*7PgH(+W@h4~bu&SS1twQ^ z_c{F;x_mZI&1DHTG-&S+opoCpG(7GD z<9QtLN;zw6po2MvbIwgQ6(4ZRz~iEY2Bzgem!h}!(w-fixMh_ye4ya9eSJ8%g}H{| zl>`2pp{4gd22GSh3zB35CVabuRICy<%NOCvANyKP`xN%BHyQrg`IrNu#JK>VSe7-m zp3o)Nl0;og{XUB;jafskIOBo73=)l0%WEY;NYi{Txf~0nl6sPc>fVlUV|u<#nfVCt zg$Y7meRt0y^B78_rzbo|i2*%kOIJ~n``rc19J26)e-uiP4psl51izEHQQhca6+m*C zLwxy)39C7opg6l4KMEUnh5OZx+>@%e^?%ji^Zoil<}}a{XS^7oqeCB(@6z%~@@pf3 z1B7L^mo%$HpIb*Rupb;-PC!moG;S`b6vx%)f}v$drPbyu{`H~E@E#kmt^18ie;o|X z^k9$_<-`WoL~q=d4>>t0)w)na|29Bli{n$u9>z-H2y==Niwf8Khh0p>MG3HI3cIA0YcpAQ(1d6`kca?5j6dZX4sr3TCGElQb09g5qxVcyE z+-RNbZ=|QLfpnc7nzKqv*-hbN4K6a_a)Z?q*`XIdcZ)fq66kxet3*16A|~$5mkOkm ziB0)SR?tXa2Y-c9Jlk^I<>}qa>sl|eoQmsk@5@SE8+w*b}jEAk%u$C zOI)&B<$(&+RLBl?s5ahb27jC@;T!W)!TV$U1Ey{q`>3M+x|oOcx3XG_l>eHo&LirK zs2walwBVBqB@+V9?ubORfZUuI)dML#ia$^8JaPp@Y*;+pfSqG_@st19r#>A)WhVQ^ z<_9|9i0V~@(cE%SaX?|d4Gg37`kJ=EC4uV&VE=C(FH9laJi3FCl-=DiJ^rM_CVNic zoLQuB%swiYeJ^+k>vtD>i-PR5A+>@N2dbkUYna*JM4T?;p3e?KI0ejBc->Q!B7khJ zTy%$F!#C|)KUq(Xb5vA(P1cYGN?jY!}$7GWCuYm+&0`Yhc9k0SCoiHp7C z*Eqp1JEXZ*TCp0S=44YFY> zvV&pdbh^mo1RY05HeBwY=ngu0`7a1Hja6?Ja=0}Z$r}u2HS-~NLBV9Yi?_iGhMC@EJ((n+oOnS$=N#AJ!-0qOPac2#`q{atXLR(2Cl0gg>)+H*xN!Yb zqLk)WD7knT-eie0=&w>J!F7|es|>KABBOg?UnatG4k=&2rTUn_p1h7H5BX*_OM&DB zhsDNR4HbDPk=1b5EIoo6V|Y{6%;}L?20gfCkETDAbBpA8Vz$K>^`XX$LNwhs+sO3w^PmF(kVQ#EAhKlnBwtv zr#-?4m*`-A?x1|W#Oz@Ag_Y6n$kY8y#_X%Bc+2fjZnC2xbIfLrkIR@xhrx#Oij|@q zsJgRRsh}x-@1YALVIyc$hu9>Uhdz`Kl)?>V1rfI(fqs4S3aFKV0K^6-;S~_xH~Q=w0mahX;Z3*h(7Ng0v8os9`Me@ zBl}e}DfFRZ)SRM)7m;8jG^=Y<1~o%jgK);U6PyV#!q0KMUY4HZL#nws#~c*1;1cQ| zB=QHWY&vtJ9I&2nd*wL)QW?F<(aF=u=%{G_F^bR10$WJHY+aJHHe@dqV-d3PfqfqH z>ZmW=mKiy^M2+UiXbOE)A6)C9@4EcyG=>wIH=m`KH=ouwo95f@gBCT_8{;C)5aLdb z4G1k8BXsYmDjpA|hJqGxb}$AWfSNQ-cj` z;|yKW^?N_ocBhPV3|#d}qPY5f-FIr98x+qHc8t<%D3#N@m2cfeSpdQOj2U)m3WQ|f zv~R1t4=kBo=>JNs{ro`nHsnoGr&In2DX$;5N6h;s~dhv za>P}iObC$;?eO1OXKJ$~Gl5HS*T?WUT_7SaEKo-nh-OE|!|I2whd@hS`BZq z(l4nWJzL6tp_pJ_>y}OYXxZPCh9tdX@Ld_5$JSW5f4uQyxEP&79yyrEL-YhwU9XFW zuMBGD?$*?_*bH(aFFtYH_Lqhosb1nwR*E0@9XM6yIXJ+!{(+K9oMeOnPrJf=qv3Gq@L@f6|;i zs2LV;m#N9G z_LQfa$W-G!^bOk;M0I$ZwrYtdHK4?a-dAXT>?xKJr=lB&1Mk5-$kgNP#5d^=4E>8o zuyRVKi$S07dPu?(vI2)%m+cTeNpYaK)7YEmKXFVn6_%VKL&1VlJGNZ|@Pk8iLO!cy zY|5SHd@6yr@03|4(jp+9DbubpxAAt39mq9ZEff$Z9~m)GLFhF@Gv{Yex$;B{bZ$doTL84CU4RFIt-`} z4{hIaFX1nsW!K45Cp5rq4fD}7j%aw>HoyM#&7IPFCfnyT#P31Iy#v>|{5c4;y^rtR zr`C^NvN;PelN3HWdGyZ>|FM`mDk@%v7BAg0jVGg=a%^`9QZT5y=X~H)cPt(?JK9Fe zJ84xF3!tn~!@eF{qw=?3?;q}XBbF-MrxxB?*dCz4z}^T=u8YDDudx-L&4Ej~Z4|k5 zFKy}wc6qM#-dpr;zANceZSZL}H|a$?k4^krw68Dv_*ieH}cZ_BTei#LuvQ{P$?bNOuPT1kBT(r|=FywG3?U-P*|%`>}hHuH>g?EG`q z>1YmvwFvjB8G_qp+rqJy&pFM$-J;jbD|Hx})!G{1YBTl662Sel>V>Ya)FSgcVj<05 z-G|DBh@o3`Ua1m`2n%CL>G903D;-xKFifUJ$&Z-d3s|p+7qu)i20_zrUCz;(cJ?FuE3#f_nRk?btr8r!)puC?jg09!w@~1PKH9CL7uN8q zVC;8ii>8zXQc81X5w~@MbfzBQR~N<+1x7{!ZI*d_!@q$1VHLG&kz`3`8Vz@zrWE(C#o# z6m0Y=+re1cB`%>YE__uX{45!x<|NAcIbN3FD#hMWfEVi)Q)XN45xLUgGkP@du7^2} z7L8TpF)^pCx3_mC2P1mlK-0Ibnv*Mb>-kdH0L&L zn)X>EReZc+q*eF{_L23cFuP3K?k8np#K2|Xj9Ly|OJI*KHE64FvqZ3l%khh5{9wnLN<0m=|Z1OPWcaaC)q zrw^50RrB+P6OMTq<)XH<^*i6gQ_fK24qYMeH#nORdhCa$pz-$AP5o64Q|l+vM>lUBzL0YVd{l094IZR< zl_|(DEj9&D>bYxScr^B|Kl?$-G_w|4LAC;_78H(tD^q-7mVNR;*Y+_5qtYc5>RNd5 zrlMQnq@!VRjEcKF(A!#otDiR~=~iN2iq6DTdEUuYkEJp;b-dp#_KBg{aAwWfWK$%J zd}~I!r6pF3m>O%))(H=t`pzDGvA+srmH|dk`!Tx(Ll+pI5bLlH+Jf=dJVz+b4YTjX zGUZap%}IOAXLMAq(JVptBsVM@sHM=o>_(DEl(D$f)$ilg+;hH%!7Q|qCEsTf@KZ3V zK?^<}yE0OCb8R>SMslZ5eltp1LY~k)XV+ z1whya=N}_89cSH?n%}Sr$Jk>5H*Y&>4SQ9?Ot!_M=RjAyGWyQd!Lt%VBKP<7;9-H* z@GwBB&-Yia@++_M&rIN{5!+FrAW3O@XJs+UI$uHzOU_Z)tRCHn6pMLL_Nkx?l!r9H zj|h}$v$Da>5BcYP`C4;2za&m&Q7=A29=ondreF-P_2;YLzN}@X&980Q6Jk91wqq5d za%dB|UZ0{1mS?k^K7+dtB@T}6+EsQFBJ+#lzXzJcYuOO2Ak}o8g-3SC=Gi4I8*oEQ zgelYdqzIq5s%NFX;R)&{(48HLAw3D<-U)i*dXa@yL*_#O^JmEkV?&&jfWmK{A8sSr zQmLVZ&Ev8pZsQQgm(L%43uy>x`iHPNLOExqiU-z;KA->0$9^r+C8Ge|gk{6@1i!D- zLjz!hmUC%HJnM07OSs;Y>DHbeGdL=aG;lq~JTD;72vHlj7cjPOL&qNi&$ z>{b3|_yb-BX|)|s8Pt3F8^6$XHZly9NMTHL3vS>oqzXRFc3G-+98#Dj^Rd@E z-U`9oDkgBKZ-~$y9?uBUXCvG&W! zBAm_SIS1Xlszg=xFV08q0?5iW*bIKaNN0(Meb-I*QHJB?yt}JiLW(?M=p5@4be*s~ zXXwgUG2DxoDSX*2L->lWE)QT;SR}+7hFseQnJsb<i#%hH)_riQokS6u+>Q5rrO= zG(?n*NTl>gj!YXR#Mr9YW0#ogPyrXfrLHxKfZ^L#KZ7z3ub+j!Wzv)tfv9f`rI{(Z zYlKiYgYHZrjIxWR3wKDhgt-R|s^#TY9GBlQ1QzGr32YiE6Pgb6We8TsZRxDB2fItL ziN_6|S32|{!!dBCwW-FcBmk3PUCWTD>L3~gC589xio@fwB&K%M>1Q1wA6o*5ZYUbQ z8x;2pg4tW_YTH#3CaYUlLk}f_Q0pHj)7)h;cEPglt%32#Ny~e>Vur{}ZPJ*kIC&AS< z6n0&Z?3~ssnst0w5zq^$|Aw=%1v#Kn4}OjdQa0=$4E6a8p9u=_M{3#(9OdI>a3ogu z-D>kI`s!THk$maA$5}CGeAWx4s&sTty5QF#hdVJ3V6p%AFvb5yFxssE)Otc2wu(C% zjmf#r=J<)l{~p7E62;Hb=@LGG$Gm&SSo>}0_c*OtRZ+g6#kvPo)Cl-pFfVHbMH~J_ zE?qYLZdbI702g?&X78JC+mlYkk(z~g5kDxYwabNixIM`7Wym}}IVHVm>npTHPj#;+ zXWSrjb1FBi+qt_(KeK||n$V>=QCPDA%7I6znG~ji%2#Zq={k>1 zFL?}Y_{zKBA)FH9)OO-rlP&jm%Q#eJEoAp@0)oqRo&b%1J)wIVJ}}b)8Mih#l@K#E zYukx=dGx{*)%)IoEk>bZXT{)}&|M%1!ur&`rjfyBeM73v@98Rhc}Bg{{R5}q|MUXz z%CR#;Z1uXLXfkc5JEcdA7Q{?K?E2Ntt@@R;Yh7IU=7seyW<99lZxfgs*wgCYmRxkAr*pvb7=s6wtcSwnRPw4lIcq(*sl%tg$Q& zmlsbU_QSKb*Mg%BFg)o;6m~Fr*oVcyFcWVjBJT8NkQ#4x$!5Gg-#I9R;S7J)<90A* zqMRh5yG8sGs-Sg4-J;*uN=grFDGqxDuU1zZE==TtpqIY#vo=BtqwE`@xv`UHxqoIg zoA#|{T6=YQ#uAx4S$CXfpu{|Ev@NugLDwFwXVxS#nh>ctpo?t|($V1RC>r5d5d9)E z=5knGc4T3GvQrx2jos9vRl%r>uuWr<#+M?7t8Kyjy8U@>S@}A=I^7)#P1=>b?1cm7 z!4Kr$q;@0HqS);gX0x*13?*Q=#HK8qQRs+}R`o*d~w znp6i^*K-1G0m7T70q+uoUz%Xer#=;SbJ4HI=i8-CORl^DKuam7#0JR#wVVlYu*LwfL{>DNPKF ztY$HC&>DtY4|s%-HkDRp&k6sg9__KJFqmGjdngGIgcVPc24X~{3eR*j4|F6@W9MxG zyzQ!=R==w;)?lh$EqS2u;j8*6m;N*2dV@_n7A|G-4xWp|Zg4|tH|I#K3t(rub>q65}o*mM$Qyvh&7*52!K+|RnWjxy03OjL^@4f!6H^@xSFbOoD z&Tn{NfPg5=cs2SzfD6|;!^P$Ijrtv{ZRhQa4@FR3Dx|r0oBZ(eOk6b`C-Yy0Z?n2i_bBOeE`Zf!x+XAuk*v$&ZB zex-SS-`dON53XOF0)2mH9@V|KXA+pMGoF~Af_MIk$&R0hA5=8@O+dm-n~v6kDYf{V z4d<6VbOjhK<;EuXiWIt;+}0=*W`qDvVC*KOdv`<@xqr&8nKW4$f5hZ#l!w_Z=IC|v zun+Xg+vR!ftXg&O5u5ssT$r*_{?U!(fN2iCFYS4GQ2Qv%qlk@RDSP!gIt2m%PE2o3 zU9Lzg74X_fcXE4K@|}}0ikzNBxxftY_6EJYFWamIV^+Jw`ArNjPF-d^jamMd#d!qn zo!tPjW8Cm)Z4d;srY*&GbWzf2rokD9GeGnn(naOfQXV-JBZI_|TKyHIq%)63YeEjkXz8-u24!CBlG8aDd`J6VfI*l z{1{DPqlWEi<`!NJY>kR>pnX)3NQPeajkj?=Gc#48`(IKY|AQ%P&jTqC**G<;n-a{g zjAs{oql#D)K8uvg6wl4r?u>!sIgUc-23AXt8E3&3;V++_UR??8HpClZ}Ob#qQ=vDE!*l+74M?3#gkwZ1OeTYAOTZDsJTVzw!-6S z^F9%5WuII0Tj&+1&h)JC=QN8p@CV!SUbtaHGSq?85hX2riO$<3Hosv-ouxe>k&)`Y zRMj)trtY5Z-U8iYuJC2%U>R5jV=0aT|(DM==W*F<$Z^+$3#0lIZ~;AoeczEP6#SQ307Z}~oWY$^PhX8Ts} z)^JnXAXHkq$GwpVIAY|t+?I!jCL4^w@1n`)-!*=3tYXv3>w52g2fH$v&8HJhY70aM z3V>%D%gaz)6pfP-aBubSp^Kc^@)!NeZ8h>x-r{v=c~YX`$F)kryv_qxm%Q(T6a!zB z%-6~w#J-H>Q1hfxa&h4VnA1ubf^Zv4n2fV&;4FSAbJ#2Xv!#gL5IMPppJ4Bom^%jlzhtux>; z?L48?H7=6r%hsc|INl)3mE|c_3<*%stE~#ezE5MLU6<9kKYl;sQ!g15*ckaY0mB3t}tIGGQ|w|3nTiQ~Ac(6+QZ( zDw#cPPzx*MuxN7w5fj1FKtBHG=J5XN;lmd$V|uLi^K}Wwe_Q^MZCEFnf56J;Q7T8% ztrx?y{XO&jPF-`s{pj+UMr8q|9;}dxH_1a%HD3-=7QJags-N_o{6|Bo*1=c3 zR(qv=H>08w$oWEZ{5pc$0alH!ako{!osXSfDcG|d^(9Ho1vS3Q^g4U=buCl|1CfCS`8@ir`O>83B3EacPN<#M0B16j>V+^#A*4r#W}J9T*;4pYXA0l zKl#BE`2e20ZzcEk(x1nd{{1#)fFzlF_^|%lnwfUtjzqI9-w*1nTzXmOSMUel04eWnVVT|Diq2yC-p);Sh`H z{%~-8KAWu`+G;AAtsJ^hyuj2eKNa%;=(b*JZ7%rV!yo@bY?65g`ynb`$zUU)GGJub zSBIj!6k%Xc!2oNi@@8c`yYV#hU}H;gN&P&0W8tO=mkX&NI|y##m>V7ijQwid-SVXN0^?P$zCD?6R_e~t zbAMr*&Zw)XTzdj!pe0Lmy$^C)c-&km^KUBV8^l02s~!M@h03kOi6_McoBoD&duHFsR|;TfZX>p)h{iI}Kua41L+V8%}$ zWlIPV^yLNGx&CzP1%Q38qI^fQCM^2gP%|ty2d2+Ob3IHGp{I{n}FO`%)c0PDEx0Og>+Yy-vuiYIqxO3=DIDMeUFKJ_Agq4MZvXy|Q}= zZQ=tap`EMA0{(ybdb{Izz>nO>sI3xUs#teL0~HvDDm;mATlx^_^CMZj(W!eG~ z$LIEPhbnYjGj3LK<2a89>xbxId*J=MudhR|mmc0EHj&3yDJzMojqk`R1#T9_{p|ZZ|FiezJ}=MNZ~XXyVdgjASnFEXy4G4__7}Y5>twvV6X%Nz zTFn!z3#)?R78iKt%{w|mR{)xc<+yz>$COwxoN2}XxIBJm{1-GC0zV70i1t@4myfjj zu6ue~Sh5Z&Ra?zAz{vO9qKIE#BYfVEGjG?yyrzyL^i}NM=-@IuIaqOP7}<1>)pL}q zQ?&8jXk(5Y(f2*t(qEl`F>A>aM=Ng06%9wr$*~QGv$I0;Fh;do1WrbGkMHqjN^AE- zEldp&u30YEHTTCR%WkcKE9Ttm+|jf$L31ZTzH)Ph{-kmdI1Sk9abvhSuv*yB13GrGVAKnw9Tz{%s(w&Yyrz2D4fdyDb8) zs$bsGtirH^QIBRegReqxy@JLh=$p1j>?b_n$wfq#-rYHhH>n%0-|Mn>9p9#)~ozXL5x{X6PG{W6r=;rdL-3XDD%6*Cm ziDP$jI9osS!S*DR)Us9Fzg^J{vo6Qv{0BX24A@vE`}aoY#Rs+Rs%94*TD;dX1?qB~ z4%b`VR3qP8S5(w*70hmFIe)!}LeZvj7-;U8;%mDuU*?kbN?APPpsd~+=C#{GZO^{i z1*1vpL{DHH3~awU#b;)5?nGGYdE=N+?zp>*@`<@u*&_!Q>$KoLr!ISR7Wp2b^hXdA zT|>JahR2R$InWmnd(flBf{-ZS;xY_-fv)}Uu>;r2P)EmWNL zDQbz&Oy+F3Y}vlny_i=t?s6ot`|+YH9hiec??f)2&BO_3{?W%xO85ye;#^+C9m4kT zGZD0_n1_*1g*}(&L%bZ)`KeX~2P_I#sjTxKM9hP|UlRI*{Z!{oELG`=BbTQAwq3u=Q6R)b_iR0~x!W~9F690ll_dTA*V2?{h z9w~5jp^RE=dsHC&GR7NR-b|~<`ZfHmmoPr5lNO@{iE2a37cvtER%OoI#6;{-Y;I}R zEh3U(4sabxuRjFWOXJErWGYfjN|~tJwj4=7B&BPf>{8wG4n~9Z*WA@|0Oz~7*wgc% zfd26Zj4kng`Mzr_{;Z$1mt(mYKafw=;)c(z&ZCI*j^-s!w3FXhQ~FonO3#f53^IO+ z&t2<3;3)n?TrkplyffkJTj!mx(Cv|X_HJY`RUa8lHpjV_w{ultXkhxg`pAWf!{eO> zL%OR_V$BbOJ2ZRlw7OlL;B1%=7o2`x%~d#t$DDL?X@@P`p;wMs(J|a4Tqsjq+QtdS zak|>R@-<{kI{ZFZUv7PXw7Jh!_v_uq2{k1l)1l&K-s>rZd&Py|mh~z%y3>Ib`yLDdj-}bA{3e<;_;1|6|NM%J z6P&=3=S8UGR($i_gtYyIyr*UEpiR%~e$yE3%yF}Rgl54(;+5J(L88MKtQK6}3-?h+ zx%H{Xzp}QD``eqz4M$18g-Z6b8Sb@r!x&33Ig5?W)DvWxZWWCkLI0jHW%Tzqvs0gH zC0NO}@lge313I%ldl6t>Y6`c>TAxpv1+?>z*pw!HG-n(&`L`ALCJC5&*FUV?`gqmW#cx&~Dca3UCW`%SH zUBDzEqj!+4qKn|q-+v=)ReG<#Gq*vf#ha|hFBji(%cpFKnL&tZm7>q0&|X)PCxzUS zb-A#Iq~`-Jo51R(pgEmYm_sN@WN2=Y$8(W^K^Qizgxr~-(vU6={otrx^6g&`Xa9E( z{R5LL81~=k>P^AkhQWs8E&bPtNU8}nWwElnRq(@Iv?7BBUbdRP>ScL4yaD7!xtEni zi!W%mkj!&?+tTgkWa1Zw6R9c4m{waQX4mdMlQ?|6xFWjt;p{BmXN0c;2dx^rW`R)9 z5ND&pgJP~s0&(!TA`f10>+mvAlG2zY%;w%R?L@LW7z5Mz8n=1#WK zdF3bJ*DOYuDu^Ui%uzJj04Ctdvrcj97D|Tm{(kfjAYX3J<<=JujRUUdk*UScPnBig zjjzbOk44+C9%las0%1~3J{^41m6pH|47QBObc0Pc&Nx2`f!zHY*8)Ig}u)Lg#Ohh#_Fu$mwq(Bh4zKF}_= zLi=V-ccI3zF0$JLRs;^p(zkl8Lb(y%qN%yZR>R7YCBn)Q*Fa2M`66jUzDmE3dH7)G6Vo=$X9&d+$8dZkPOXhcpU% zgJfL4@7y$$Ems=|}f=p%A*B}`zvawQ}hW;2;Mq=-tA z-|oRZQB*=PzuN(QtnY3Uy%NtTZqMvsCglLSyyrFYu2lw9U z*cO$;v|Cr(9jLFBsjdf|?U>s{-SLVofFb>fyj>fj{ZOl#=mW2bgYm)$QB^2EXJp7URf6U{*`45*cSWc4qJWSCpQ;^q-n7}CMoQh;{^EH?% z57vvnOFU8aunt-aN_?{$;d^M}9V$2ggEv&^uQxjOuamd&7&P#i5T4i^%*ZkTrQl_0@xBViTxV5m%b7E!*?5+cbA)x+?0H zZMWvpBhYI+W$<$e|ke!Lvpk8-7r|?}9%?`WvlQ>%2Hpd!8rW%`CCkb`?9X zeFcrzE04BEI<^EgawRpF2g(%sM#*CJUA6bdN5D2*MOd*99?;&gsYS>AmQ*OmhDv(X zYo8qJ_874*hTOR*_8uvM&|+6b)Llh%Tia=%Ap@5xMXqc=Z3uw&)MhZ%2)kFAED?To zRmbZQzO}M_;e!X;B0-+d-fzI|BY{Uzmd_F#^xhfMq?1rw)lX-6hf?fw{&6qiNiA}$ zs6%9YJ$xF*A9C(^^ye8M^gL#xx7Kr``DR^;r-G<8xI)x#y_1bzZ69=~oB4kCK0-W1 zox4=e{`k4Irfsa0$36UQQWjYqJF?dM2p*LFG%`ok{76`%Wm*)Fn*B@G%$U|O2vbA) zlqKr)=$9kh3qKv{mGL#ZzRb1J;Oa2dJ5hgG>cgeDXORBR_^4V(=y-Rgre1AbU(5K# zwf2=B4<+l`LS`ZMq03y`tI={AHP z2rHuWK)2oo_Pe>dNaw|Y8;oGjGav(=;QmY2vY4X2Z>#t4uf?LKsCz;UI226`($~z! zqA{cHiFWK;9=W$VNIQAYLm)T3-b&A^hnTUL zxC}BSx4Cq?zsSdBs{{fu-e8i?1%_F@*5rFTS2zv!wQi-&KuTXssUjueW;$mZMFK(G49W&Z)!L5{_)ltS@X-jhcJ;@D?V){&S| z`l;sZtgz(3G;!GvZNzs2OOax zcFcQ-OBV5_F)TX@h-;{w@%@2{nAKT{eZr|`@euYBS?_#~@}W1P3b_P>a->wKoxIJ6 z_j*85po~Jz2#$g-H*7=~IftZS#RIk;8*{;~Hot zfujk4Yk4tBse~pSl>S#zB#Nee44k@di4ZM~!yWEeHOSr2_T9M-#S)S^I@Zhx>E(3% zw5{TeRi#?rg^XAw;|0P~AOQ@}FAs}-0l-12S_71kuqFPgaOGwOmp6;Z^cSm!sZVf4 znH3pz>8;U2;Ul6m7Yino``Fz$8xA*!PL8&Ud<%9~-3W^BK=B^DR4fyhU2Of+<=MSA zjt}jK`$ZWQ`fWV?939U_S+>s=a;N9SiA=>+z);yf{gFcf%qSv)0x5mV{duAyeHZUO zl94SL%W*Hu;AOb5lp!#mDcoC$YgBr*R`;i1RxBD8mWRZ4Y44kY5v(LG{ft&>seJYpA=G z;>u6a^hDpu+jz3lxI(GDJ`G3P9TL)Y;?O2%Kkc$c!UU>PhVN9Tp$b}zW`2coO8yDk zim_@;ZA7*M5RN_J{0?Q5N}_XdCf8lftbtTBA9V2D$}{Y*!=ZIn(VPIjLx1O%CTHzh zWA`g%@!>63Cj!Hh{lS|3O>#6$i0~k7vZ{ftDdO#5Fa2iHo<$a6qItbIbe(TxcYQ{L zw_DJpZVX*az3@KIsSaSfO-d$mthUfYu)r`DNwwOBCu;!2%a))?EgHu36t^ToJU1Z% zvy^9;j(!6;Zl7Oo=@m=5w35uB-y8Rlhl{Oo6QS z3yikS9*G9)!)Nq9?wgD;MTxH)&>&>qwQhWA&RkVzLaq_xlyQ@YfT5k3!7> zVlLdB?aUPe_nI6xS3_mZj)un?W62YTa{~I_s~$aG!bCB-ItmqahZ`Mkwwnzn2Mq|f zs?IwJL{Df63rf3%^;`FV0o1amqDws(ov+?E$w!9(b&3s~YE+S-m*@I_O;gc&Efv2~ zz@bCjKs(Ux#`9dGz`0oisV47~k3;iVQqyl;Gol$mrd%aY% z=C75SO>7Ml=QtQBccydm3Uz#g(1cnaL1l~EjiOA-)C$d;@0h0p1P<1_D~179j9eI` zgvMD$opA%z)z=*w=6VvdQhloPh;3~uz^b_>qtJX+o$s$yhZaf~U%iR!_u90+iMHhn z0Y>8qJyY)6mj4E*|MORlfY=q@!b&cc01h6QO2%EktM&7x3|-sr3t|(p%y?J|-_zAt zehQa&&M58YD&oop(3?xb<~9}E->qX?fzl%s_E=-d<@K2tnaIKUO49^gF(Bl6ByHi) z!BMfqUh?^>{><}9O*(_%#b~0q;GaCB4B@>aOk+{Y##{g$f?9({@eE`bkC*ojEc=III$ z;$*0QF8sfyG5WH=WVz}m`X^1X%C>NJu@o_W5zs@2P3P|*ZodIcSi|->j$R=5!d&$A zJns$DU#%o%9@gml@FU5od-ztEtxt+pgod`KEtZ?ZZ9Mab=v%xmv-3-3bO?-12Mv_r;WRm_Ylcc%t8x)lhIR_o`#9VT zlvkh^RrTO}-grk=XvB`EcZF3Y9ovz6#N2*4Hq$D68@VT6`sbg|0_pKx+cG11v4`6K z3)rwqx-Eb;YgXdsn=_gyA`&1R8uaN0{|(R_#joD|M=yYX(n0>v6Radkw&N=nQxQzY z7T&#lI~l#_!!N5fy#m+j!7@9;i0MI>(2Q7v>Pce19;~=d{va$r4oVPcX?IUo8t=KuT=J_c&kej8A22Oyst8z|!B`*<7eSaG}ft zneat^L#zb=cXu{|jtuT>K{z}^}r%c)waPcdj%nQaNObz;7 z0mE(iGcFxXp4u zj&{fa^W5?1$#|##@`+6WdK+{u3I}&Hr?|hNvBxhZ;~t+{*>Wd40*7*AI^~7ewLm`^ zhnSDw`2qtvtW%}Zn$cOD3*ra`b~GFAC{}N1S-lTmN#^L|{ubSq|FhvMh1R02s$a^U zn>Wes%-noNxxTDxUsonOHFOnW#G!g-AKNP?RPYk@X=EP+0Z?f7$RJa_DqrwmskEaG ztQG1i;)r@%7PufIw|y>G=vaElb)hW>Pk|g0Kx}I8Y^-6u`5G_6mwf;b`!rIc>AV2_)+nXJ{zx6Nq&2n0bujJ*7FRH zatVZ7T=)!dW%(s*nbxk7xbIFq3!!x#dc6zD&Tsn*Cip98Z_gUoE-P9V`Sc6I zZE_GAJoL^)r(pYjeXEKP-H=@eobbm?*yI5G+Av}aOt%U1V?YFdOqAj3oxc~JoX2=7 zSu`#NhTIOuzuEZz^{bm>a1NB*V!xeAobdMz#$p;f`==B9A8ywNXJ)AR9n)8(e}0;^ z``c?SMLJ*n5AXZuH|tA=%j@hW=Y1O8$KRJgQnFX<-<;F`{VN4G!0LTt*S7pmb*=yQ z@UR(we$4-{X#d@e|8+Y5_a6MuiTMA_X51XZW&+#2y|8k1Px3z;kAJSlLwW2NJ;yc{ zu}_lATns`gu^kUjT~Ghv2i*Aj&%|$Dxg`#Toc?si#b9UF;79fGloy}M68+6qsL4RX zu0=tS^QYa}OAI$j!DMG$Em!;R$NYEEP6a7I-F{N)MnCDO?n2*s@tUlPl@#HXOToXe zq=?e5QFxLUe+hZQcRPaxAzAJB1QQb@o$*pN)0Kehw;~y?850CB-3=zT%HEjs`mrS9 zRx0MUvod}l*7ItVqf|}2R76B%Gx6ksq<;%8MWjt=soDcc3~UmWQ~%?C4j=xGq5PHV zssH5n`~Uaz{-4+Qzw7xg*5&`z`Lwv||KP23;D}ufM<^xl^;(dGuG5zeHu#>r`+#}= zQv;gpJsi1Px}2|m<)LH=K06NLuN-3wTYt8~%BxAvy>eBOkE$}XJr^>XUNEOn;NJhP zbN_?iBkO{qnADNBSIp1Q^d@Jk)Oik)3lP4?7+xnwf6z|_Ohe@QJqm8)yL#!sc#5%C z|9q2dxvod{pYNbQC%a_EXJAc4G(z(O{oA7;ptEMUX=zP}aW<-;I7Rpo=Y+?6@)~66 z3a2f1P=j1H5dS*tLOS&p0VNNUP4zhAHObGkYuSC*tYy=FeWf zu8{5dtEhh8>Ste;a&ys$by1!1ji@l$o)obr$YszO-oN1pF;CYRCs`!ZADf`BY@NbY zS2Zf(EeCCkkSOrMq+H)A&6B=zVzfnw@83|NEYz*AGwK$a&v)OyFeHkZ>a%aT3#!gW z8mCSo;5ZBC;<@M3)$k}Hm(=t60#9KRf~@iV`I5o341W$NPf~;dr~Cd~OeOVp+THvjXX#i)f{|mP0dX6DYC!J~=Js zmzDTov}qW4b=P3x-+;aPcmGJ%pBb~d9L}yG93N26B>pRcR9qR)t zy6~FcmvS3Olj53Xe%3CxR=-C$1+O#+WJ`**+@}IVg1dNoM6cK~WjpRbI2dqzbbv#r z3_ap-Ss1E*z6l9RJz6qgkR-Le}ba~{vYA);ryn^~0>JBqd*WY6E`{(yivIVG6%dXL1(;pov) zn7mZX@KoKH7TmY6ExVkYy8Xqx*UG8lRUU=c8LB{N90YsASHE42D(pa{u*0k@pr#tD z=2NIMl~d`^yV)|>$=ym*WtSEqyqJB{eQk0S`FJLf+~9eU2P9+~zvZYk0SQK}Vd%Z} z`16%#0cC>*HC^wx8V((Ir!M+Z(7&~w20`StLpVYg8KpyUQ;{2TlY2!F@7o%v^@x;d zL?U5Z9jQ%Y%-jE9nF2B&@6~lheBfFQ_zZxTHLe&#-+sM-hLfXJ3`Qzy2Uf4`rD0yp z1t-v{5^9-`1mIn61dRsCb>$SA!70#C(8U0sdDG%c+D%~9%y0S;OZsAkgzJHLk%73<}l9z2Wx@GID z41Kd@i2_<8-w3n3L zKFWSa6*bTe-a=?o(uZ;OdC+hAWG+MXg9GJEa;NW1OW1idg@Jqon!_-S_^gppg>#}> z3tk0XxMrgs>&@g%kYA&pNTpEGC)4 zKLmz&Gc~-QW!UZ}4aya2SMY62wCwjrR^kno&i*^&@ek9Y&La8w<>BquuHZjflCXQi zsBW>Q-08U>V=Bvfw^wj5S&TfDHr(>D()d6U<~gQBr#il7y)Jo4M>sCwMs#EWlHDNd zqKwT^k=#Iw!O(n|x^p~O$@X`kk^GUud{-{*#U&jdMjN;9mRV&B)%$SHI7>LR} z7jv>4xlBb1E=7W|1rdWoJs#Z{d?)U;INeAt`+T6YtHK1XlqA|!v-n1idRLEi2hoAS zUx;jUC*T7rv{mwQ1zK_TZa&4U9FKGyl!4qNRg+K7pF1W}RR~*Yx{sjg;=K5p>xtQ3 zSQ_)|+XHUpM7>v^S3wf7^~)sMV5eSZ@My$i{&DVF?H-7xh1?`+MDsV3UB6pqausRE zb6vlywGRH$+kLrgaN4=F@5eUO+lcp(FLL>;hg5RzUA?9!traExGxp36(ACqelJBE~ z=JWSC%G7ru4MEzv4hluKgxQPt-NPL&s~)x~`$U~MCmv`&m7lP-#Ogc%xg_PClm+f_ zlK>}6iY2I%;l_$jnP0sYt4EB>No;&wP$rwFVcq9De7Unm?O)%@%uCR&&Lp1rOyMY$CqU8k;~^$@solaBH}BTN1?Eyfw~?zJ_mt8`2rNA-8cmF zj>D+!Dz9+zuI`cSc3)^ZSOzvRD9yMye@R6MV)`St;VJ7ek`J8&1AY*tOOTmh2JX@A3#mP z1#~WX#<~P@f0EPse`VOC&fJA($|G;^2GDP(RM8Li048!z0XWU$h+C#7$M&Au)E z=mU~=OpNMBcLZ5)VXEf{)4cu!YEpr7GC87h*)Q-t)LR%MV|DOl9aFh5az8wXwj1}D z5A*e)C$Zk^xO1=7=V-tA3|ll7`dH}gF11P=J0x1UurRb$=J?1&M%tn9#0ufMI7F!{ z)Iw6&cqpX($>CdhhTB4{`WvFlerGQ{FA8%rI}~8sy!S5DlE?1JbF7KEjz4$xRluFR+Pq#J|DPB2b$-fGXIKwezT+XdG z{1Hj?(;JM-G-~(orgFdug>RL#O5_ehD&@EuT0fGcnLDQ{^omUZw|ztV^F`lSox$N{ z+hb7`j#oE@QdeP%#k=iY<9?*j`_%y^F@7AsgPrRAB(YNh!S(DSJzZX%s(p$gb_Z-{sd1S151aRIfR$tyW}s9=e`tP!-=pJq z{R*u{C^ST=ujdeRZImAV&<**1+OxWWWWiq^lzV-%!i=&+287{~e8nBg$n{(BI41tW zc;l|f);G1YMvPP+S#qz?f#P8OIEdOV4y7b0!63K0B+#&@2qi}KL{Ckz-ck*#N=7XZ zKV~1mf;-@e{7nB;YPSAN02TJdwO}OonAm3>r=LDW!v?{PgA|V>P>Gq zuIsOod8Vjb8B4ot)_xAZ6r;^J`k(n4xa*cjhY!v0Wr@k&CeUFN z*)Xbn-XK;r(bmqi>A1+FdYdcb(A>2C=UhS>)vBRX@bDRizN{#EqlQCC9sDk$#hLTG z4T}#G+;Y8CXJ2hKK%OuC*^FUzGINq#hsLEJ;!~eGxEe$`8T;CUa6H5tObK&6iq?hF zzh6VyK=mad@V3zQyxnD?l4t!JZHGvUaK_0{RQV5b9@isw=y?#RIK08fe z#zZ4qu`w*|{tJ@G!J7&<<1QIhNt3VYnKrtvfVSv4O5}j*NxORNr>D7jSPE)+&iYwX zeTFT~{1MB>uX}bt6L`6)0V*yOm&8*BiI^ln?>LQo*B4uDsyxG#FKh8~N1RkvP4GP( zhFoq(37rKFo%_SYRWdrpGacnGdkAj#t4>)3Rj~XA*o{6#DoTExvaDtZyCJEbS0P3; ze8D)F_FWAE(kmYX^e!Z{nRc@o%S{5FAQ^mFWr^j}@l6*uM7r!5Pe8L+WjUvb305jQa`3H8 zM7;EEzhKsv>Jx9}4a%eOWiuzKrkJX;ue--hsQQHw`r^}M`Y{=mlPGoh>K7jj&tAE7 z?_R^9G(w^Np@q7nf`1c9BvahMQ~E=O6Vu}0iO4H=j3+}J$nig#)i7nY8{*G*wyNnT39^pn7`FY=>ALDm zBmOy_5*(^N!#-GYuD^B=cb%vtgQ)y%iD~(|=xxzCZi(lKrVmTf;-)wa}89+Xlg_33+TL z%9xuhuX}AO`E@-a?ZR@BbP5s1DvNb7Ej=5mWqa&zL(j1Wni` z0gVygqP5+sQQ!JMPEO=#LSb!cLFd^C1DoG&U}N?a)osl(xd+VCNLVa?He;7}Y44wX+qf%@xUtJ=TmuNRk~k)wMdB{xe& zN%)UV?8V2BV~yQf8ZJ$vh`8J?UqDC7g!dU;{a{IKdP01iQFJgLL#{C);0LR1DV5HB zqYsDe-U_LO@e6ONq6Pr*ql3Lqw^6i_=?OIcJUZoh(~cNzjz%?tjg%))vwqLNJ)PR- z-R&VzG)y-|3z8=kv8ZLLZyk4PwB_;fX>u~eC-0DaP>6Uy`iBbbl@E>_bwN{Bcr(QE zKk9C07Z1hy`nV0u4roJg40@3pqkxuY$%J-95rTMXWY#Q?+cwp$*$4FZ&_x4!vAx5R z<{7B!p}23o6ozd z(WM@4NIc0^KYWXzN?;?8e59wYfd(OslOLbXpqWGFAku}BI2E&N1;kH{NtqlbbMd5H z1@lYVS=Ysy`Tc1xvdXHqy?cPs1`vf;6igiVugbN?^RfEKYF%ahDsn18>TjZZKVF$= zP6q}ty5Bz&D?Rwt`HnL9q*nj)hq2XE%BaodiirzZ$*wv}g~{zP&bBGG@b%p&Zds`# zB~)L>`XamW{8ev9tqfXj5y<*LQSn#5gZi;7}2DJ$JTh$C@ zx|82)QGF}ZizYqBIQA+G8+<@{$bIeEdRsV}nd4sJ;6_NPdhh>g>#~-C#eU?rzmV0s z_xcH%U5HPtkP^V%_=*h!S5LCkSQ$LJZVd($o3yr^xqADPyl&muD3-(K!$&o|oaB+| zBCfe}KxSE0-86q~I(_sj_2iH_h!Ib&_ix()@>j+rpbZ>oWM~}jW;2D7O$Clkg8Kc#-I<6cAQg#m{pTt`g3EY# zFmYn1MLq%qphYD}P)sCWrcLMbH)aK>SUL~a4H<`!(#%2$2d$PN7Gr5&fwMk2%|a*+ znnV3y9v_^Bbu|@YO|2`hXba%jly!AnS4yHli-IW>GDdgg*$TeRd9=7dG}_X-IM%$4 zZX!VudhL0VxV5cWi^yr>==Y^(hOzO95Z^zOUkXl#kj`Ujt?6RXuMGcjVwQ)Gti{pM zONCN`S$F~m!(Nue5er(4|8+ib;?N8&CSfb#Vc3(%R0TOtQ&1ogqx2`03AFZ_B0CqW zV*I!6TU>+jkwFP!w;h{Fn=Ix<r)rgo+j!Uo zS+knz?)qoJ99~te+70>U5HBa94hJ8wY@%zM^3&Q#uje4D3;Wgrk26<3FQ^_s4a!C5 z1Mt#bJtTgQT0`)o0tE4#<|?etMyWfR9&2yIEf9Z4+imY`u2eBs=e2VIYgq06*sytC zS?2udFu54k1t?&83?;LgqE?PK16VEl)gvD5rXK!a=2maB)pbI<-@>i0>P~$>{eJvt zdu5V+E&$LVuS!Nl;X}2VJ#-y=v91;!<8E}dr&7})sJfotW@r=yW-9n1b>v~HQt?Bo zo=kbCQBDoZwYL%C`1xn6e<~K#fupfbqP3^;#eedE@TDKQK3^ZKz9AG5J_i0HhJbAx z{QNpAMm1GtH7)6$#c-ry!DbosCHVAZ7EMRD#vj=q?yeg_O@K{b*JC*UXtO*8X@=(- zow~)YKq!BjP4m2R`-@{V-)8HpTWOA{C+9iiBxG-pM0UFJ)}I_7JOiI!-i5;5C|1hh z8_wn|_n6oiKbd@bu@Q=rY3rfP?|nS;%B4du=o8V{&Q_G<7170~GUpFWJAX{Gnb^sa zxkTc^Erk%fq6HmKAe3zS^qn(M!n{@ScHwkl!XW-9bnt33kCLwnkl38CUl3EO7zq&G ztSMI-bK&)uuO2e1gwU9+7PPougLD5FmVO*%%Qf=q(C^WAjd!-Kp@rp})X_?omETX3 zKZrwM;GZX?5qzv6qxKcGu<+jZ3Nx^cie&9yx4j9 zH7k|>E!K$)o>u3BYL)#)XAYy2dizWiD(^Un*&7uRB*D{uN-8SSp+u`QfrAxTr1 z*(f><2#>~Rulv&$I>6fzw%YzyN@BBJ5D5-2q=Wc3>uH%s@tXKQum8m9_??1kX~2IsNTkx z<+-Skns#l9rO|XWi-<2hWSYccrWkx3)`XTpwN@rk&nq!|#^NGkt)eIy6b!lAp~6e}>D=@IP>*}`Vbl4JqHOH4%(XDdYa^7ocS|^Nv54c4z;wP5 z8LOd3u6?x4o#Ywc#+BtsQE!eKL6c1n&Dk1^!mamxKBV;Ay5dL`O{LZ^-S`T!bsorC zb;&5sQJ@KyDnki6cP#>>Ep0Tl8ul>cQf9o-9iM{0PjJ;=bU*2aJj{CO$NA{xsqzs! zYFwvNSr3bgzmv9ee5{n@Fa@+*gzJL51AVuHX77#kKPJwbTq|;l`&+Wvsb6g&7enre zgZeKoaIa2g>a-$eHz#TW_Z*rTo z{9t{qdL@$cB_6j~)5}ieH4M2R#@&{UOE83qlaBd)1zgW(TZ4wFHMTnZG+N8Z3?icH z8L{?C+RGh@Y^ zFqcqvgYM0f*|n*rv#$lHFusCO5J#fJ^N2o0K|1E7$5M9ui6I}Ey5*CWpAH1@B7j|Pn3#Zn6 zSE1C-Lfuy(wLra}dxX#{$B(IKK9`)eSC#wLttS~s4Lph7GPl=$Oij>_7_x;mFQ!UQ`diQ zJs-G+YxdxE_!?FJY>82)Rcoq*Xg?;Uz-`Xl0a9u=uWs^=JI`$(lgv&L(~;ULzglXi z6$5s=;%5n3R&2EREvK0Es(!aXb@7E^2Zl76YpMhZ8AN2)M?Whm=JZHrY<74mMN{nQ z_0MdoF4U4xho^Cl>0X|+De57MzN+0iTYjy1?!_Q+)Q_W=^iMdqspvINp-@QE!N3en zv?-BW!YKV_oku7eC1EuB>y4$g0YFH_YP)Jr$r3rukkbBe%)lqlmj(9|Wcu?r5nY&H z3wEI`2uoluR^3xgft#lavS)2>(ReH{^6mPRq)rI$px#esFbAgqs z_B}xtZC{J%=au}G*uLc7{$!S1BtL=cETmk~rd~+HXQ2WBoP*Unr|C4@8PIFV8h(*K z0Mn_IM?Dfc!zu$5J)cGfyFh-ATP=i;;f6Tf{+RlzY1&VpcL^KLjx(Y229G35 z?hGYoW8(8!Uno)5Pty&W z6A}ZTy;(`hd9=1fBa(fG>eL@ya)$I8-u(@?@9!NI-$Y8_=hS|@Su1!see_SHb4Sa z117LlX`JG={pltC#PFc^CwKoi-EsOKn)rPtoY?_A`|h&8_=kU70!0O8Yu0IS^PiMe zX}7QmWXtC~oeD|+^wU3mkw0(RpWizz03MWBP(0%ll)|51QosU7_SRscQ^B8qdWzk= z226O&hjo@ynW2Av7YjwObOlx8N!EYy4{xA_U|8YGr!+(V_y&^JXdPtN)|?*O|7_}? z4)DMJnIs-O=(Wd4`G2+I|6)`Adj?PL011`E_}A{$cYQF#AN za*w>4xPKUmQ<*<~Inc*}a_yIooAzHa(ACfVFko@?e-x}3qjj36{#55{V1~{y)$|^W znf@gW?C)hIly!YsQ5}-=7e4UDmuRr;RFO;at8TG~zheB-LI6bY{r_VlU3~K-pr70n zG>1CRC=*Tq0@xzduWHbQog<_o2^X4+dk2Nw-&}OU^guZ?e%3K=wr-3HmHJcy@5s*bKqT+m#L=X+W6jd+b5; zTU893^0{sf2;lZHB>}XS_$Fw~vTBYIlKxcKRCTm@4*cHnn2H>06DP+4Kru2SdQ6UY z^!piatUuY_S28C?7NM!TU&|u`kGX>tVOwP}p?intE(psgcgBV_oMZT1mxk6LunK_8 zEkC4t-j=e*biDpQE}y>SBy^U(2(l=zpTJ8F0ZCi)fw<;5zO$15QWdi!(GNx&2x){d zOQWi+k~0oZgp!v-D8nw#sbwjJMQn33?41a^RQCT&Q+}JZ18q|SAgzGlfEIPFfuO&u z`GFeuTCIBw#pdVgEn=bK`W+B6u{8DHUX@`n4L|UO-m)_m4ZtmUyt3l;{ ze4^}m9|JqV;J0(q0Dra6`f%f)P5O^2+pg zKr%-tNMIv%vmO(lgxunnYyNK1ZYX8}+ZP**w(UbD^8z$mFjY`79<0es9BH$aM*>QY zxix6cAW}cI4Xyh45qY{h`ahG{zHnnEb2A8;w*T<~Lx%I)oAyR@>%s7k%ab_`*#C6)-G!_LLPtjSr+d<93QOGUss5`#O$>*=H^wb_g2%dPwh8$N|W}7 zww_Gnxz(V8Y=my=B(0wUx8PmWYP&oToS`6Tk=#UI)p3`ZBo>tyzuio9t+Ewe=GWV) z(#jE!8$nV#-tz47A)Eqxq|MiR$xv50}qVWB{S3 z-(7g^CQ#F(a&MQSCZNmL6qHzpkcOkRi{`1p6${gkVC@nY*0gxOAW6{FETA+N2 zLzY3$moH{kbflgxDk&e^2TGp2~&%D!`0{o%{Vw{K7x` zelCz=dZ4XOznFzHZyhkABvkpYBl_7Gi|r%l@up5Dj;yg)8|BpNGy7yzky>rgi^z83S7!1HbKq zCT>x$ACrF51KTJzZ#QVXe-pX8_91o=EzI7Ac7j@g>(DnS>{Xh)yiedIOgbpf@B}kM zsO|(R3G>>B!&;zWZ*K*SyE@STFl) z*%vgMhVEAq_~J5ikT`ES3Q}sWy9>&`R1(UyK@Q14?yS#>qK{ka$U+0;6nKA2WjQ;# zMP1dcgO;^qIBNP3Y2%pDSWG!8(k{~9q0)x5g5Jh(NJ*7cXjxQqQ%u2hR%&8FPhsuL zor}$u1t54fLgyIJlLO*JJsi=AmByT23ik3VtyJ#=g@d=or?ytq!4 zE5Er9T^%`g>UyE1N^l}{HAdGCPyOK0363(=geF0ck8-&@O|-;bxUv4bP;|3I#sp_0~7-LN6@H^Cw4a z;;~Bt9MB7jmxCwOe7g(X1-#O?W!nZC-MtuzLw7S(Y}PE-1)tdyH1y8MfnvJ?Z7eG%+UH{?2(9yrrEAZy9qi3lG~dRdB{95X>b=co zX@fI7k#1YkcHO~n2Q9dCN2~JT+|kTC!L_q3O%6TlyEg$|V>(dZrfBFGonxDV#lC=U zfjb)1U#|jkLT%bcu`I!5mn^A!>|hfeKh;{w5EZzuiV@yhjbxJFsW&?9W#P!kvmwB!5Pn z?*w&j*BT@|TY}pTIj3u01>jP7RR*Ha^lp6zf6Tz~Bj$l_piaht zcG*(X;clx$c*9e{fmR0ts)=QGu8vPUC(xd=6Ok-D6>!=3qV(|IBMCx#pUW8`2dFB$ zlD_Tozgfku8!8=c%=4yuw#qXH@tcayAzeOa5my&(uK~pI(1N;@@9aFSR4WZ9$Ga5s zxq{_;rKv~kD+g;LDJ>51gaV^C$IT{!Rz0ZZbyyVcV9SS}UlW}7-zzw&TlFsLDvsa9 zJW-SMug2??m)w99(~ECi=-hQOKapBL&?# z{>oNdOR-Kj*D#4nTCJpTtFF)IZthPuy!=-U->qL!kv3_~h}K$FzOj04>b&bO z%OvNSj`)`FT@YT8+mq)qP^QMM(;K#%w1$VdEdmff^a&g|yDJtgj`{9<4U!bmxcZaT1y!FMRQ;*pHMcrFRRk`l%!cx)= z(hY)0mvnc7luAfR2+|?l0+LffKvKF>I;5pbkd&B6NeU89;=3Q+@B5xJ&e`Ma_2)On z9BVm5U_SGXUtRHF7qEnm%)j~WdQS+Yx-h2kXx6&~y_e>Q4K3gHxGaMAJV39Vvx5h5 z#k%C54S>sb(-}zD_P<#=8W;|2jT>iOO?PMkx~OZx&+)i6A9N*!IZYyYdUA<$j>anm z$p7@Sr#_Ti2|`&jxh#4%qf<^%Z%OlUSJN=q2V0LQVI4VwxuwQ29VTa`%%$@6;9(meE6Xfp&XD3T9|Oph2e)%DO3CH}jw$PV z?~!XgeRFV2(|hBAY#abFIld2#oAZc#f<ebJB8zb|j)WWE6HoBZ z{KVAHe!Kvovak`~rIad#E`1=f6p7Y9~Dy% zENh}My(Uv==aC}F{NIvo*>9raDkleO)%_s25VN{K@0j`D|L&Adx}96q40jEYet+Z!eCnQS-rR7MGv8S|ZULTq1WKhQ>H{C3(g zh>Gzc0Jy2dvxYqyODE5#WHO9${+YJ(sLq{0AT}wq(O5wd-quNKg{nFmyhhI!OTfh1 zVn@A|udB5~l(9O6WPw-U8j?c48;UPUUa6Qh9b}ch_3nlc_QH7~MB%^S<*g38F69ho|^d!(Pa|*OS)SZnSj0H9k1}|=c?FezK=N)Wn5idZ} zG54Z&DW4ZtcE|6lIG`jPbrI0A1}TKA(ZZD^mwD|T70(aI+$yQ9S}HuEQM5LvJnr+p zIMT9KPL2uF@$uXzzRqslx9Nv*+F%YY0;5&z}oJ+ zJDr@NV(A8=32F&^?|ZFk{onoizVwV&R%u||Q#w{AL`qj%1kO0Um&mnoFdNO zbya-%V7p5x&W76cGecqwe0-P8n)`N6iO{P0sYD|iV+K0hk8bk{bP2QrgC8TqF8i|C zIX3xi!>B~?d#d%KvGd_M#(7B+QX{>_)MR=2nO^66pPCZ&ykBqPD1UhRFq;o_6%7)~ zb~AM0+Oc@Q1$=zMjqEX%T_+Rxw)~3o`|$PwEt~MbH6i|xEEeQ{f;=YFfoCntvGK=d z)!G)-F5gX!R{F{yf|aDO&ijlpYd~a<$n>j(};xi;j4QR*q!I0Oi~wV4vo6CO6ptZiZa7jq#v-Z*y$PN z^C87L6G&NLCIEPl?vQB?F}i*KO_9JSycby4>+#K#d*n-k&L8O8w0`!({n1EL9%au5 zF)t6Hj%~k4L(#1}y`&utFeKV%LQSnlqNdLE!CG*vS3wP*tS<%tOIGSyQn{K=1x%l}#prgs4=FUWUY;h--ku;2+_5MDkAWl~&|%f`X7xiB3CyZRnm(j2>a|{K z^zxH*0+m=70(z^31+WCkdw}BndMUQ>>gVIrn2({PdDBCDAczWs(P<$FMh=!hd_+uJqSXW zc*5bv;-4CA{m_3djiT8IaE-4SJ)T!um9^ZTssTsfqxJoAD{egV<&f8JAF>1nj1fFu zX-PiN3AmqsaJqY~!Gd@+7i1-f;2D-O$8I0817JYU7dJ78k743F)h|-W{>A$!rw19W zjS&4n(-R*zwYDT371$M8oR}*i%JRiF;}-yP6b@=k37(nKW;-xu}RIn7P_Of-TBZuWhmwTE_qklUaCoPjftZ4B-7^g$$ z1*Pni-IGr5noVCEWkw5d%!1vATWfNj1hU;%7V@Uq$a26?^Qh^)&yUVCYn1!!2BFqhti+Dx!x=SAk5P_5B4T%}tGQTsMosF80psH^TtDniw9W#oY;k%i zi=Id$!XUy!rldjVU-r~*S#K$L2ZxS_%fgZz`e)4x*nfZ?4^_8XvDz!2kp{Sab}fb; zRl@o!_N=d}Pd+Yp&mGd~ht$eA6}$wEr(VZ94aEh?^%Ed|ZTpC!(Y%yEAHXP_Dxy}m zP3RL!-eZXUq!;k(LDx?`(_F$lEt7A3=vbFZ*|3tH%TKJvTI8&P^(NHxdHL3>o`5xT z`bsi~Q!}8T4@tV6?7V%=ifxsFa^b6<3=GSDm)Lp%s!@%9k=1AUJ(wlg;y~_2hTY@% zEJ=W{xF`L|)h*u@3x6MI0JyYhkA|XLhvi)0zCj+H3t~px?8fJZ`rG-<-9pN8I#qZG zw;;tkAr#MM_iip}hVFqolcdw&%l?-2g%0>dcq;T6j)bdT1&Nvld0(q(tRC4*bSB_Z zR^5yURZJQcRCc)7i(d##Xp;dDYZf|H3B%n%@-bSq&rJnwxQSf7uEJX8bfEB(3!>!avO)cWk0@OGtlc zlQD)QQ~lU%JHP;t(Y7UrZ#J^qLI3jc13T0II+L~7!}ij!*WC|)_1Becr&FD6XS_ld zse255;FhMoV_c;(*XygeF9qtUMBEYUB7vK%q&Mu^X2A~E4DWl{1og2_osUGa^pO*a zS<6n+90;-FRwv+^o}xg6+P*9~D`JxyC_s0LZ4NvJ$;ZD#E7ACt2S zaNkB_Y1rQF1vg5m^zF4+190B%L8ngvZSAs(6hUR2F>bB+n%{xXe6|KNdM|8^98iyK8-ko9`!5Wek1(t^j{D_N(OB%dX-zSHz9SG zukA%>ljEkw#W32!#v8qb!daSTL*dzTeF69#QjLcpTpSl%d4SsZ%$N3_dm@T{!ITpX z7wT%5Ehpj;fMbWBnorhD*M?Z-TqQrA(MU7=uY7428PJVfhkt?TLCkB(XTvG8xhP)t zxXPwLH=)T?T6kYYhnTBoGexo4D0eO+LkdmwjkHjsk)1T(ujbwhGoOV4r*z`YT1_TY zjY_f0018oeTfee#5gZg)V&4r((c?d(gOX)|>DFxH@xnA0eLjcWLPOh3O2UNo75e-X z%t6xUdPW~6iORw3AU#Np)llSI`J>nq4`gch?@Gib8zj%cT#g)>nY<_zApql89QH)OVzK~}#0_+IYPuodDQ z%4;HwwJ@WAQV{HZh{z0t-?>(o`;ndGpjrg2XNUf!2jeM(Ih_WM$@2_eK55< zwZ}rXq{{+(?P0e^lTclI!7Fnu1s7n}HIyB)k697~E zw3yNSqsiMlu6y?JglRc7Lx9*zg8y^!BfDX28Q?@!tJ*gENKz%_QCjRgK1?z+#l+@r zbyg7yVa4f=XRoMZ9oMY@lK>H`&k2i~42Gv}U1`y#>3OU@vfl+D83BUy&{cuv@w-Nk z?3i!{ecuq^nurl`n$Ugel$I!_A(^2jR#8{@MnpCXq*%8U zjkeQ)ME$2vb9Mgn7xfLG(;Yvx;=(TNie^=UJZuSKoVe11YCp0`v=_hk;YYHLFAO$p(ydCkS2a`oy3CJPg$u5utQa`nu`l8gj+n|gT zCHlc_JZf?Ap2!7Ipk7qR=OK!?8oJmk7Pl4pS3Tb)gLYkPOzmtZE=(?T=X zE5y(Y7M{lx_b92-k0sM;Z>Ic7n#PR*y)#fk>E)phr3G`=rhTL)t>QiReC7JrjhqTj!=UHf&*($+G#{xJHVOw(BFt$jQB>sAbkh)e1tJr zN2aFlZwsQjixg|ZC91$sqjw=6Pn!P|7&j#SIE^37y;t|yEzB(doIp4J+b>t@QnSjXOSR%Dh+0^pKKJwukaI=ig^Txp3Sox z;izkvqwk)Bs~ZwJf~+U0W650JPMusB5Q81IZb2>^=pL)#!6J*Qr^`OQMJuLnXdM{! zTD2zjstYKvH5ijE%(1@DaP0S#?{>EN}=| z+BnAxS2zV^i}P_^HBM6DcQ&Al{TurW=&AaNTo8^9x#8WF;+H|nb>C;ssLVb`-F#+y zIg>LkS5J{JY!7B8gCG$C+xT)xw>u#`L{c!OVC@}LaX%{q&J9^Q;B4z==z=IwnqfKK@7^UH7w|sF< zD!d+lz(;=_Pt3-QkwO{Vspk)?qCX(5;IM403+=@(ftY%2%m&vVnsKD+b-liX88>!? zk25n+YR1?|c|mFkjX;DCE?k1C4B70sY$jxMVG`d(t@x?+$Ba<;=JFVB!k$5xkT;=l zKZ!3DKlv3(Td_&N#rM?&IlL$@%$vOZjh8{8>#fPwDzRf~ETjTchqb>O^pyEJc90fG znv0+~e9Tm85%t5XLd~gW( z3_Ce;VST>MTx_6%w(;d$lSy6xjCjup&}A{@=AcplL^}?uEELK60_;x2!lOc(uv#-g zp!0QpNNQ+Ih$p(0?fu@r1OP}R9tdZB75r+<54j9;f*u`1iWeT=BsApubjmJxN{smR zgN(a=d%D!n8(1JsJ|w;2BoipImdc89TMOS~iCxBYFU_Ddd|SB3gn*@!uVz25uO1t7 zimdP?V0*se_$OM%5we{vedeu%#-ypGb*kp4E%BC*sHX;nG{p;E>@;rKgOn>?s{v5C z6G7+|ypN>f^F*cp)hz3fCuab#8x9BFn7mx9OO6GvLth46iJ!g{mb8`TBf7}?JK4!e zJ{Ud4Q+9P=(H|Q%XnXc629N_!89zHV&uWm#I%g>o631~*ZT*UrWrK>{N>J<$?26#U zZB}b#QErw%5+1X@%<(!nQM3WGLRGPAez&t6M@OteldKek4HfRYbKR9(4a8#Oc_VEo zCM3BK6FW?nuB^dozMa;XL}kJ#Fhz^q^GmHD*!ao$-;h-B$O~S)6|qy2j#E`n z|FRy$gx+i$wxT;N(Jns*?99&3A%!*JdxjO(D8#7LlTmW*k(k`$!rO`sqXGq;c*3mk zNpbD(QKfYN>~ks{bED6daAmGY7U9l2519H!5^qCWw`;Y6jsQI>lIa@EyW#kZGe}AF zm|0c(MbrcMg^=(Ro)_@wzRmu1+|>Wq^r!Rf_O~=#740_tG4|2a${Xn9v0sVOb9BCm zg}YzqD6+xEP64tf8M}{%n$@{!Ez@AB)d}B=HH;Fy!3tYc+mSh|b(=mC066>==---ATx6_TG8?o_N~< z|DyG{334@G3t&RH89Vd_P#{9K8Ce6KABwE6-v9+mQq6RiH3^TlkxdwZU`Kb&G{szp z6%}lGHqLoGzE6^B`d)498|2H1l`b-_w45VE^cYX<>PwpP8E2#g#(>=M6ubrLlG56a z-fRBpdAqqoX}|2Qo565t`?)cjNn@U#a5;gQ3CJ?#Tr6Q3qs~FWGo(U%>2j+(6bOQ2 z%K@yrj_CRY=icS$QjfWahc91;U-_*W(u-V`tb^NU~`c~hcxN}Rb20+@Xg2poe!D?En7j)OLs~# z-@;|2BP3|z`r!{GtPzY0TbD;uCl%jvgX% zFC7o(n0&gkNVz*0F1rU{j&Z?*B<>^Ew}$C5xk*%A-Z#rku@@cM$~B8`ag~yZf4)SE#{Qpx z5j6t_jPJ0EH-A*`X)WV}^rJP|{<)qIyk9c*>oJW&q|}m)n4&YEpr-vKr)j+oW-JP_ zK*&w{`Z{BwY%D#?GP>GWBTs0biX^gG-R@;Sdi!NOaiN}xrlGA()^YvczUu%Vp+jDm zF!~s_prjjQ;V7On$^&7q$b}q?a%A8E2J>{^rD;3A8(=EeNUV9OHh*So33A{!^1KE$ zz@r}ll891>SNr~#5{ou_$fU6FpsNmAt+wS{qix*l_y5%Z(#-*4SPRz$~@ z$lw2@G#ocvso@NiG)4|5Cs-3-#OMbEAjqfm_jm_=)(Y8`r?Y_HL{nPKW4wb^O9+%z zFvIqRw?1P|?oNLjeqHW^G{gNPVMWy*RSWZqkXpsim$+&btNKe(k;1#2^pgxgpz=%E z;Z5iHj^)!;dlBo5W!)kYIjCakS+*6C_NJn^OY>AvFDGY*@qm+3hPc8Yx5lu9@)5+A ztxD%`@-aJ>a{}IoIFL+*5tyNoGaD#sBz3Hz@LyY+7WK&Gt5X6o$b=y0`J)&ut84gE z_vLnfcpRHpTrpy1m>##~k%sc5vsJgjYvP?eC^rr(XlsA;>!Cyg%NN~TQW z+E$E@cHO>3kMQ61BB(k5SS!jp@Wu}kUEqDafvB0nL7yeDR$t=r&68|7Am@X0@L*^6G5Kt<#co+hO$M-+C&Brp5v#RaF_!M!=`TJm;N zXwz>0*|eC#uGCS>TE#9LDa#T90iiU&)jl{AUl$(ow8Y%Cld3}oSNSf*zY06;D=!l?SxQpQLbyzDs#l`(So1C>B<1UeQorflp=KSO|9jje!65S9dr(s8RZr|pI$xrO0 z9p1Q8AAF~DV~>uzV>5&`%WNv`@BM0a9n&FvlOd0{LgZOi=Yocd9%5)@9v;FlzU#g9 z*bi`{<4C&OyNq?}J-nxj#cKTh)olMC|pulKqPc;7xK_cx@ z5IxJL=oV)p_Sf1yFwP(Tb@>sXUc!?M9^XUMH2~E9UqXn&H_E+Z=UkN6U?PNI6S)I` zr(w7B-Dv^kdiMJSusfq zT<&iL>{CEQ+~TW2(w25kYg+U{Qp}Y3--Z1j%HH2UWk1Knp%UT{2j)+!kcB!q=ws}1 zx&j{cDyT?f`}rnTW?2-1qkr*Cee5xQj5Y`msell0ZNi5wCm>oiyR|)n9Lo(N&28*O z&=53_mpJ{bk(AL_nCVkuYlzl_MCHds1)gs1sUK^hG#SYXdk>_Pb10mXeW{qg1f6?G zGWbGEaL*rbrCaRNdC}5jgI*LiDMy6Hgi|gVnE-t#bD-YAeNq?y???+Cr1Us1fhIXx zob?@hY)G97AhYBEWdU1a&k{`2rT^?HWT)4szC{zky^30uS9#QLz>a#d0+8eQQs^-Qvie%?Z2}%|XGA z0~BfOdpPPA1QcT_Nl^}CoT1nR(G^-^7YIS?YLkGTAOKPe22OrbP=N-ivIJu_ySMm% z(NPMA5YlcOMQZ;ipWx5c%z5t+XhysOy1X0^P(QQ9oB>0g3nVS(kgV}Zf+e{5{?|st z-LX5CHw&_Q+7IP6PlM2?J^N z)7JQZM7nf`Sn#)*fBfG620Dse0KBdL)sBSf0sxe0AhSOUEkxtrdiHNH`hW2_rC9+! z8uWrHoDOUnVCXD+4{0TAfF^qdRorhoJU-Tty%&b;n23`}W?amG-K)a=Ws zgd{>zEueK8v--KY@)7GlA$SQ?{(I=jJU`?w@aI&+pY*@`cSr31zBc2nfA)#cLCciLUuMI@R}o{~)ORo1gcG zivc)rtp}Q$|8P+#|8@zF%HjW!S;PN)Ot#Rm;iOGS2NV0__Lg-1fA%SfDralG-b5+2 zeh)}SdjOY<3DJxy9X)QLz^2>LsQ+>e{=*HHeIKLFN3v?(;|rj)P9qEdhw=^oy$3br?=!jPr2jWx|94kKh8Cc|ZayjGO8jT%zywf@tlkA;WBm6! z|98LPe|rw;O<=SzcQI3}miLd|sPsEB@O|Hxk&*wSU-RETdmPfM^=$XpA^L}>6{_z% zFB*bw{cmsozklw(d+PsREgyO-nqoXHji%(qI*-F_6~tU>9+f%UUxElbcL2$#k>}~B zpk~dX#D5gU*v2`)RY%o&7enbq<7$KlzTyV=9T`Z_^YgzLoB!#}2mYOCK4ARZuABkr z1cpW6U7~1Faq)XTnFiz{F5ya3+ZPl?@>&pfrv;6(KH$LnIZ1?Uz+Cs zbAkWocey_XCl1&2Bji)@{vL$CaNKV=%K#Zu+vf7of4tqsagZVYEM)6oWRozrZ?Z9%Ald|Q3?5l0opaxK;9t`DZX&%f)GyzAX?I&*-i{Gj zMirIGSTuqln&2@#5)rB#f!zlSBpHVwB2@W%j9AD>Z@D-bNQk9XN#rWeCe@^6f+JlC zCrSgae~k{!0!FK`V}qOK=!4A7^tQ8|A11z)f*UEzTR60_RS1bafCPj zmhSiXtHmOnTcGPwl{ixA_bhEVMga z1Ir$F$WjggLzLwx(lN+uWC_R@?u5r4wSE|L zm(2u@7{EvZVN%(<^DLuNil zYG4a2IBrr^VskY<6<5nrkVciy*21Lk69s$>8pg`mJ5L;V8^El)3DVTd)Zcb1-J{v> zd}kKWqy_6_?L9;~LVGA=*@xo=MsA|V6y4U!B=cYl)T+#zKI%J4bGu5JB1+Lp6q;Mb z&QbS+>Da`%@o9v7Bcv9JcNb`Fa6D=4ztr;IdDzs|+Z^N2O^rUr5ylF>YUWt8`SqNj z&F+`T*Wub}yqp*BZ|UIf5x`n3a%pOSQBbpq=NwDscPU6IG;;(j()4r5ffzqB*-rLl zD&%&LUC?IrfUXcr(y*pCtXijgg?&8@%MM^X`|Z_MGF#$t`#MZ*aX6L#jU+COGo)Ys zf{I5n!2b>sAwlpn%!I~6aCvmN|ALrLFTnoiGO3`!Ua=&ggJ0MM+&UW;?puuWx1;*k z9;8RUdMC|)gol+t;G0dyEo9LL$djSDS6M&rD;FRDp)oCHDPdjJjH(*@&-P=GPKO0B z>;lIiPr&E3+3s`%L24f`+FZILaF12$%Bn!Wp$SyQUkM9YB!#qpr!XgQv@JfCM*U5L zgvs3Er(#h!CfX_LLB8}YD+4`MCX%aYlE_n$*9=)DH}7jR>egq*_hAl#B|E!79N`6! z)jT0}27&Ol*%i3o?;Kh5gD!u~&kgRz6=Wbj@b%@Z*N(cqVY>qLD%P1^Z9h!fW^6n)lYZgUJn|e{Lm_(l2qoI;{t1z9C$?>&))X>KQ`GtDn>83gI-226s#-Mz1kpKFA?x3>d*JV=g(^?a9K`Jkrqmaz_g@YWa%erll`44H z=N2B#x=7Wud~=+26TFgf`CnQ9yVdtGnSEEr(V>+Duartm0jh&+4VB&CdT}Ny;1}4} zG5PH7{?olL{xvcN3@@xOe&@lh48qV1C z%KSdzi@yFlv^W{4W*38oi$LIWyB_p?&r-+y?xyb*2L38F^T3cgvN_TiDH4t%Z~^>W z(~~=|xe0qAnfm~jo+J?Eo68^nj+T!1ATV{kDeV4Gq9bw)TAe4P)kQI+PfQf+CG`_A zOS*87?v?!nXz>h#Ot%rqCyvV_b(rJ^R7gz8CG3h%qDrhVNHz3^YA{Uk#1EypZ{^8F z#}$vudIGbrA<{PyKf(^|#3X)iRy7F$??`H`00Zmype=sDX>FF{Zl*n# zxSZpb7$!57?TI3E1%TIvFY-vAm;(tG5uY7Z#Gd55A#?VGCd&(Fkg{Ba35MD!s7G?9 z19^~u&nEAc#F7Gp*Jz3ZM4xC@g8SHCZUwYP)8cgq;qqUM^*%+CGoE1)8DxQr7HOH4 zE7)q;Z?g`-`oe=yVddk`EtXzj-)U)+lI!Yr1T%tDeT^*#dF@UflN3~#qDrL)=8BaB zfZa%ju1{rMyH`$cpXZ|hy&@^Z+!)OIH$17VTX%FYm0`qNDGGZa+;V|1(l%v+k>ztwmIRg zD-j%vjKR$M@)(*7-(+k%{e=YhLJ5s|sHtR0dc^pkqqy}PnK{Q<>o=5IwI7CapVzDS zspBlG=ZhhqEHah;wk)snQrA=FNze>s50Ph=0Qh^UlQZ)ktKt$*ph})ZCzR?dFkLiE zewvoel)8vxO-i*U9p<_?RE}lUs{YNUv>=mcY_q)Ey#iB?{7nK0D1g1@Z#cBNN zc#(s>T*1OgjbT_gYQzPnPR?YBmTUP1%}-zX6H2@m$Avmi(+4HWuyGj+^#0DHyy2cE zpw1a}_C|NeHVq>(2%|1EXvqyS@N58T-3^v+G#pWana9xQqX5=7HQt|>ZxArM_ZwJ) ztsWYEghAf{UO39gBRyiu8_bzOgrV6I6jc7^b%DHVRI8DW%w;g~luhDCA=ai;vrc+9 zP}~W-#aje&WrY75d55!*FP(lTxpa<~Gvx_N_D#Wj8ps2|uKDiijE8t2{4SeMBzd{K zqI$%iQTO6tY&85B>x?rvFLB~|E$UX|5)d))W5H)v;KOcf9x|NMH=& z0y<>J(H!krGj<0?D;=L-#hi^Cjzz8&jcQGYzg-GI?3tcKq-5-1WAG5>>*P}(F?EwY z6cM&QF9Bs%&3hj65VGOIQpZ4t zt)zx=D6cGO)A-WlDMCN9oa3v*`&*to%vBB{O7%8bU+@>Ceb^WyhTI8QMN%NQy7a>)t@oc_dwg z|D|)xmU*r)QA6I?a!g50%nN?^Ep4g6NR0+V0c7B$QQd)0pvFXz=1544WvR`f1>1)z zvcf{5-t7j~#B`nTH64Y>tT+&5tEB{Pf!e%IPPsi9^N)w;1zZwXrTTX=&pTs$R{8wB z-H_zfMy86z)j`0S*cm8>5ju~QZJF?t6WJ`R(xnuTWSM9+b%CloQGQe5oAj|Qg_yh# z6U(xXeWb`Sl|e(Rl-BFg`Qc8Uhi~A^P2Jw$(ItegVX8!3u@?9IsRdUfcAZ&t{Tjtss16+qh8sA2>@=suH+IsJ9$Tx>Zd09|sQ znV!l@Tq;s|K5&}*u$@z>nvJ#kc(81cc258`)nJJr@NKryVt|?>lM`iU4+Z}-?+J!$ zSAEOn^YJ>3mv;&BN-gv-(*>U=R|=4#!=qhBX>eAni5=2kISo89c%!g+>b3bwbTrJxhkDiUP;q-oAg485610EI#F!O3FkmVA`s@_;X8{{<>zh6;oyk~$%i^2x>v*!oEbnu|dyc1LJxTR|JA@Z2; zvqj>fblE!?a-@A7t6z@n{)Q9TPy2ETat`l?bn*@QYd>Im&F>F7^3Qiw+fue3)(eZ@#$G*vSE zrWJBX$z4-nqpfmO&k&t90c{IOp+5gicIIQt#-Mfwt`hCj=_KTa31!lo{o!EX~+nWET|DJ=i(Gx_T22`K!sTTmnO{SHIv zPMBr^7OC~VkcB#sbShf_{RBToN8t`-TV?gglnK5e4mdpdq=7izDvw4!wWrglZ~s1! z{X1h)Wb9eLChen%@|~|P2QEVaZ()bRk>M^Ji#7!G$&V&nhzmu5)%6l^V_S?5IAJo| zOSJsJ$h;@*jt`7P-&X!q4ymQs6IGn3Q%x;XOMfFo+YO@w3fOTWI^;}d5y&bQU3J71 zA7SM8Tx9k$V(y^iHo0y3AhAV#<|8Br;IM?z zK5C-LQ4ZtrY1@ce+hZGA3sw1~>S|8;w|c==pO)bogSsJW3abZ&xoNGo*H+ zTck)l)DNh*(j9PB{Vh@3T-M7dv1t5-MXlaK0Y^UjA!k-2HE9$t^RQaymH?BDG2i#_~v)}Ln@VmV#dg2 z`SwTj*30b;Lj%!2K^AdT6Oycp2>{nSKQ>Yz>E3(}r}8cazIMy|~) zY`tBQPZ~)TF;FNbQyrX9LuLW9ZO5#bF{(3jld|&JUcjjS%li8HyV81}31{bT&yuQMfA;lG(7bHqALg?BXM=g zp~IybPXbI5X5F4U29eB%&!_2|o;kcfxW||SbmN-(ON=cem>Ip(=e^>??*~+^3x`bH zoiA%Fx$~#wc#PV$Dg7;4{Oe22hbBES%!{A|>kDMQ_YM(gi_@f0hHVz4%JuT7);Z&g zFDCOtJpqk>l_0n2f=iC#82NBIHQwXGlY|@`VoZCKsp<}d=53j{>wQ2|*&i7rfY`CC zm67$R{^Su@5(FxFaH1sR?@5(DJV)RZjqAeS@e2~nNZ*6IM-me%J(QT(1!fFxLO*dZ zzgA19g{Vg25|T5v=UvI;V6HOo8qNy4i9jv0R*H}=@P&ay z9<{QnVWYbqi!nxF3B)acl}UBr=(`BKjp_Eu6VX=BiVHLCYauGo#+#*-Pt^%*)bl_# zZ)Ys@7#j_?D7Gq{9*^MQogDAl<15j6Cc>J&6=A;b8Fc^FAi0Hjw=2+31 z$HN4!YwkL{kvVUj=^-n3GnR05NkIH z&ZI@_ixP%KnWy{uM0;QytM7JTBcKu;(Vh~8K0)x}xgji~Joqf;7U7q_H{dy_D5Bi! zIQv4p`N_9-0TZwd4=~0R=K!qAfA|$S>mE7(niX>oA2O~>5f4`n-%DA!_nN&|!UaP< zC&f&jdAOulgb>*5f!l_?piaA8OJX3 zsmG)WD^#lphh^LgMK_L2gf%);^VDJB;A-x|2$6N`K}!Z9WXYD0N25&r+Lv=7_7y&B z=1`FYIita-{$1qI#6P=j=Lpvfd4S98*4~RxL>OW>^Y;>Z%;{S~b*Ce_?n!Cbu*=d9 z+DxquA0#icBRwO&U(c9e^*)l97kM7F2ksl{fdijqSb=o+Z|{Db3I^HHi4(NB_U^=; zI5CQff_B42(yeP|{db3O^60b}D?b^&#A2MxcR>%smB98wcA`dML`QbV$)ccKQ%1Fq zB*UC}RPWNLetz9cW_Ge5W5w%pV%q@Jq#|mummel&JY`tpmwyHc>#hD2dK1LDh7&Q$ zFiVE0t~cHx?Vy@i09>%|pnLM^3!|j)91c?j%Ybeb(?jZD`gV8QWa{3!sxQfyU1i}XC|iUbQq>zmzfa{(QTK)AT4H8d3tA+ONG$-YqCF+ zs&$8+54DJ^RpUjUFW=AAk&iOWD%ojQ1Y0cH!`F6BM zoF_x z(+H$e@z4`cV+on0wQ4km&7{Piqtk`6+c3w`8_CiOx~$3nO3Igho`!LPf{cL^MLOAj z0pPK-%h-(n}P$%CO+5!!FNRLFVl<&(*I zc}8Cuu^i36%}0fCDET3P0ob918f>0|Gx9teBz`M2AQNrx{&4?b?R_!00Xr))hS5>W zd@u@S3n?;e?pbcx^C|A=jWO_}D22xLjGwo`_GJ5Mx*Ni$^=8o#U0vI&zgvOJAG=lx z1^5R!_`E&wZ!hC;`yG}43x%#J=(q!>;RXPyLTfH%hG5qqqD{=iF(8cD;?q2xF`b$!)! zgiqG0pV$1QBa-}E41U5BLZgr7KJ$MvPsevv9-erX*Vd;@dC!tL7}%|M z8hTpV8kMxBRtL_oc*-p#ri6j@%?>15Ig8RxoNy%{K+Cny!6nvcy-PcgXss(NH!r!! zA1PzNiyFM>#&asi%e-dZqcuT7AD4F%y;`ZV$THzQnl5tf1SB(mhQ~(YPks{x9nX?o zzL$$V-?z?`W;OJU!!jahWcFu8@r36-3Ur^xLpC%>o2D1E!WkrZ?JcCN|;HIML# zg;jvI1V;D3V}QoILLAcdTQV|>!rs#m-CHd9_NQMIElo;@4g`~HZOCrtIAD- z*w=OL^FQZ)aNYayf2~=rwOugoncp`)@qWKP*MX%WHK9AW!ydGCkBt^NoohV9?T*QP zMfMLnx)Uz>?jVQwN=MVM8j<3N6tBUr=4Su3<~7j0ke}g z(6^xCM&Z8ddsy?mE1tm=vw&u$+$UuSlrd9ET(U1Ee$Uk37cYw~rbG#d_igM(RGO%s z>*3c$=QNa12i`@psr*0M9i-Bv4H^`ZzDD4>v;cODG8a=+`a3+kWw(_O9 z#aa1M((q{+U4Mck?rSh0_V|!YbY%=XaMD3?ULc^psvrCLeVoL07j(9PE-ni!ISh+3edmn5n%0+dO-?1pNVn)@m}nj z0%uUuRp_y#zS^yu3WPU?BXS<~uVIu!rD-%p&f*s6G^AUQwK;>}zD}q2WBRt>$PfsB z>;k-(XU7D#Vd#ShE;gCZgc3>w8iyZ+;Z}dzgLb7LF%*9-ygc23>N^q87-h&+qoR;l zF+1|Ocu+=Tr3Lk$fD!FJkov8u`^p+=C)ueR&K>~=4ka8E`|*$Hnr zoJ&oqcMlR%!z%kd4U0o)A;=tzAle&wEe`$C|Gfws_uh5#E!UNP=f6@Gn5Ls~7AU>8)DAS+WpBwG!}x8VQL{Tk<4Mu|xb&01$TdE(UFt{CZdDnQx8uf+r*T#!7xacUht86d`_PVTtx;#4 zQZ78cL)!VZGZR>jg$z`q=50fm*E@dik~PQ8(X_{E6+9-jdpk!rNPZ~4?3%&o@$)vF zXYRZ-7fPwOSQmcz%tzqo)@m9Ii);b`SI{|5FZj+t?Po0>?7JIk&_`02G8Big4Z7@6 z`~gc<4r}%mBL#n=IWtav3N3J#wfS04yFZwNa+$P58qlg=>8_|gJD1rn02P0iqs13X2KS^P~k)O3~p zm3OSzV;W6XGtEHN>Hw1&)z7bP!Wl4VR^S~LJXYp1)oQNb;S^wj>QiDh0CHi;q?U+;_5lMg|d zUMjBK6qBV$9s`#UPigOmf_fqts(>|%~>Li zph-@-6iMzUTCIjKq3dIg7so-_cufX3)2`rS!g{r5AdwVxuy3bI2WYa|m6;VPLm;{QHHZa$zTTkLbq?;DW21r{Q8LYU&HU!#Vr*4I zVl%>uF1ky$krvp4M3}xjQ;IVcCG29BGVjEg%{ls9dlj_B-rXiO`8yBhnPkW^LDoRT zbOH@6m=5Dq-@kRm6*WYC8UxmuHJ~n9Y4&}%`{*k*0*)=8dfP}TiMrE^i~O)U|3Tp+ zy+5ZwK6AEBk11bq*b~QFf4=wCn=o;m8i5nM_^{?VS6S7_TE6&>Jm-Veb~il+^tPL( z1yZHE2C8mA^zIBW{aUYT51&thV!5|h@VgzWwWUH9Kjp0~!)!>*O@}A)|IT`$AnU~! z7pvw8?926*O#8?r8spLyEw~r+hwu?i)Q7V0L)XII51R(pjW+o?Ky_}qe~1U>Aao_> zXUW(EAK$r&aN?)?Re2WYZ3g_1-?Zfn?=V50!!K+h?rF9Fp z4LmaCV$pr<<>@6D@1zlgYD;g5K)q9Fn3B3rBF>6~?GFKkC(}yUs4@;5`g~9N`6F&- z`c360IHYv-anq@b-k^}!$M$#mr7AUbGeId@Nw@YW(e1GdY^`C_@^p^DaKv;kkX&Fn zPTkAGQB^Xd5c12tYVUi&0Dp~1>jKC<$D+sYX(_h)8)@_R`lxHXTkcC5@Zz*s+X(HD z((C}-l7VNJk0xFJFkc^ML}1Ce1B1W?drJA-K}aZ+Vxe-V zF9;OZtIL)lrAF%SRCwcqL_P9#!47gc zSoCQZ5F$-L-OH@K?lUT9Y%~v6J%Y;R2F{j#5qxwl$m%D|lEW#dGM2UEJXCGnS*ZLLadPq0C1s9iAC@A@ZVP*XtH6-M5h|(Z zxk7zTHG+51_@J2J!1Zn$9E2f>HmN-1bQ+}%d%cY{CbZHERdBmh)hEGnpNqflj3IJx zHv#?I4GcZcxUW%>Htl4^$jOYxJFQD*I<|3fSiE%72OG~0Z7|_2I-+H_ypHwcVOX*M zj`g1StN3xbZfMs65DU*hP*?<*z}v-}-DVwLgOLu}jkZA7*~l+vOU8^1#*2tQ>T z*WuI?Ry#I8mJ)Q#$ucFx+50vCFWwalF-RtGrdc}$4QZRo+XlNDI+RdqagxR6&dqOa zc&1!9Jha^?F{AgHDtnP1=$LO11<-YCeUvk(46`)g3OM2#6j+p`fvA=couxf>IxnYd zQP4=TS#>I1udE*DxzS#tHXZ$2Cf>UcInGT#C;RH&_#CzsbWj`P?lAkT6C~Knmv2Bt zBE;`5ewBE>IS-&$EsaP%?`F3|2v!yz_1q*TJ}IFdGplnOc*`>aq<4+>P|e6H4d`R_ zM}D(+NdA%vBQ>f#w@;kK-Q<1KHs8=GHcWnqMU2Wuiwh~@c|HsOU74MPxId->-ZA|? zAs~?SjjT)}I(oHbk3eK!baU4_gZNQ%T74<=Uc*54l|G^bJ zs`!R{C|VwXZuWR7fNbY24{a0H2<7j?m-D~b)oyi25i-8xAeF)rtqS-Q{0r`@d*&w8 zS*lL#e?!r4Sb?+y71mqfNL1Iq$gONCL&YDD4<#!FGs=Y?TaQ4k&OQ;K5~tjdT+tBn z2fCP2g{@=4UC8fN$HrgP z&`Cyj@gCg0Waz2N`)iy6yZP-JmAFmpuSjumo?H0%XE|ZqXYIkFpJdw~^Hls|`epmO zf(moV^)c>2=a>1k0>u>_D-MZ#~eM$|H=7J*WXA#NDB{k2oYqk&g*wd#_9C z&4kn#0frGR1T_mGET5cnV+G3&_P3KL#q@JekI9~D)64>w|dpvUL?wO{FWn= zvEiRV;@Ra4(T!MC`B-rb-FGsa7-*na&Zgu7`zG`0+M~m}kJbM)t<}l)!FcV`BELto zMvO(fKPng4ibwfE-LqdJ!>yz4>Wi`VoH!2T3Q{L@e1VU=!<4OQeT{O_xhL-4BV^R^ z*%(Fqo8j3R$$>Ith4?yq5_?5_=h0o0ci$%unkLV_KHDn9cb+*-d5lSO0~MzA-0|Ig zq6vGX6qM|t^QVqZ&L^e$L{-W>2pSEz31|UMdez**W8>rqG#9nvJpRhxH-_(z=6Fs% zd+JmR~MSl+gfxYBm(jeF=rUh}w3xv+ZY{#9Vk$bl+4WYYz7(s%z`+UtqD z3u0zg;j|m!Tw!PMWDi_2Le2NgQ&HyEa-;h=)a=xHN%L~pUs7o}KkQK+fn+J_wl}h% zcs;u$@pZ}MOzXCZy8`nfbf^t`509Lu&jLy;fE&KPn_)eU& z5?@pSvUB_c82bbCfQYonu6N&r)GC7n!Cm}K)|kBPYxmE^coDimmO{(%{Sb@=hE$t{ zpofYt?0>!-bt)aGjKARd&M^0SDDeFYQ!&w+jw56i3kh{WR1#+b=UK(2)u$Mbj`et)mgMQR+rnA6xnCDzARFI+9ybO&+F2*l&$;)Ez3($Zj)(h0i- zeFtEM$a9TN>)g)ttGw>+@=N*^b|DDfKR|ZW^)ZTM!3YY>yILzF4)QPMjc-Vu2S*4l zm}mr6RoQ;j2bsU>@U;@*e`o$O1Juy6!r>ZRmKWb$rKj3u{(@g6&(GSFcigiBn)GVp z+ci!4puqn{+3ygi*?xi*?xjcMHWPgp2Tvzx*brH*m?_i}?QuuC2KYTG5}8noa_{DU(2QtksX;#I^? zu?p3QEI!JyWlKxEpOP1+yBC5HbFwiT1_k#M65^VSjcb0rJlDHKJ@0tAth0YrtA zjpxQzG9qQ6NxGUy;`X)vj1(J!jZ1?sj(tyyPyyx}eU+r!lx#&o;4i+FrxBQGXHNB( zce***ZUZgG?NOiyF0DY8mI2ja4bJgpl&rE|km=0;ZjG@04`Nt1USf_SNH;%}H}grU zy7^MDbcGq_GeJwXT&o+dyO`WJAUCJJTImftY8~hgI8Hjnc$%aioSvpt`~IuX3?7dR z6x)RNPjdA{gu|?i3bG99;_VPqp{ z_5PJ7p|6Yo?6ty+xmLr3Yxbsl=NwF1k*x(2>5gT?swxhC>IGGzMbBVpEF^~BZ$hws z@=b@`cG+}8mEeDUICRnTKX@e>03v)Zn4j*wJO%hRodi(3Rk@qJz`Z-tpThk$6b?qs zi}HdJWon5Qh?;MMjHwu-AU)Js@6*`kMRja^4gGa<>F*ZPfA;4-mp?j=R!n))LlYr! z4ixOL(x&o+?A?GOgkKDV01<}$vunUSw1Otw zt+EZZJ`H{RCEpXI-_Po>Y<2(BFZaKmK^DHh$NtJ^fU>IxCZxY(dre7p^lX&A5+DT^ z%>Dfq)JeF(#?V~+QS%WLuvS6V{j+9sWPE6>++-Sv#D5oOo1q(Y|HIc8iMeEp3`bD2 zuGIs6r71Af?iZMdXC{2^n&?_K4T3o7>!9Zw$jn9)>o;9VWDBJSz(Yx#+uVEj(hX!h z|Kv5~g5Kn?&rV4A!_a3=D}50Jwc+fqz`(n$Zo0@gz6yfUYS)MV?5%mx2{rlFMA3B= z`gbh)Kl)Py3#}{@t(&O&$L}yp7$s6+U-uGSY4~R^7#viiu;#75|Fd5QzvV3so?O#> z>9wHnfAj_X`-h#Qz~+Vp6e<2ozV+|FM@RSn+3OD;P@J9x)*oJ%AO6Wl84^XW6v9%T zNB{7Pq#gSI7sTIRytlYv6S2%dGFP6xjP*ZVnM08{#h|8ealAI+Ye^(ilUY#1^beoR zMFd!0Zxv`JwwwI(0NN{G55f5MIYyad;{BJ0O?Rvu($K@Y>o8TF9fX!w?`*-K!pX@dEIq)H#&au5$ z`XAmR|8?t7Gk{}DLE^e){Eyz`mH_w=Sm*RVaQ@%N^dJ5W3S0nT!E2TO=^^!hzio>j zBenmMUvmKefA;AAo5$^+1eP`j^xA;)kKTlI2YrjZm3Ie@&8z(Q=sdcVy3CyNtYq(h46yCBxJrVLH8`8`6n zdAEQTwB79|r2pw9*Wx9A9p6k`9Y*nui8~VFDdf)SxgP=oAw1;+Z_X4z#5*rrFK>@fPgyDmYtAg7FRpCFx z(2~=7FrgLgA78tiy*l&V0>crr^gtPRT3To?`p1G`Zutz_#}FB0T~hq@h$4^aiNvs$ zfwv1^!Jmtr`S$4tFN;;bq}@X=Xn%9;^X;q4vgRW^GEUURs_&)6F9tBEo$Lr;1W!HL zkL0$R(FQI(r+#kv(UI#zYe)-HH#!X_-)(`$gaexBJ`F~(+SP)NPd%V7b)VVxQ)*Ri zrdxf_{fu@QzpIA){7DE4HGJs<(Yw(hl$PS9rM4PE7W1R+6ji?S)j|^_v{jrfhA#MLhDi~Ry+Eq0SAT4Gwu9czw%^D2vDqqYFdtv-h1Rapnw3RV!v=U zmu7yMj2wbE{FKq(b(*{=16E!+3x-0V&H3>bLaYADZJ+mWy}>@8#uD1OQ42S%wAg?as^4>TJEGUaQA2a7Af_SBDl3xDM`q{f(TNa^n2Kqut zSJ>x+SLcyO^hntL26Khxg!~FIupoEsp1%gqa$D%ndxFmZKm7D`8eO2j2yFq#>?QNT zj7@@dfOp>A{aHbW>Y|kj^#)yPXLM=Rn_>&JCja zu{^Z)<%$^Wn6INu_|AksexdVN*T?pa2MP86vOjyz_2y2p;BSH{ej9=KFzZ;*=8DH2 zdVc{Hev4aJ2Ma6cJZymtc95QQZBti58|j30;!hR7^>_`5Vtq#2g&TRqhqi(BE4QZ<|)ET+Ae<6k_>&+V&kA0ZV?er`2}T+#{klI zDgcMLP4abP9^E=vj%It~O`x*%^g)>Z=3bP8+=C*k0L_hn$H>-%KE~ro0!2-_g`3kQ$qbJ`5U`9BjC*SftJjqzTr)&AXv%!_! z+J@CreL1BGY~?LrlvAkjmhqGD`vlIMKZLI+!TeQy1+uN8H9?piy3|Y_y5o7$!R8Y^ zsVQ!A_KwY~^${2y6z;-*k~=;1%%gas2BfoB^h4{7<-N(#&-=K-SE;kP!nAE+zCkX9 z0#v%6wa`S78RbPX;z@L-4J-z1wIiB(+~(CRr9X=ee)Mu>;xB=;yW7#$n2zSH_EVfu z^J7+Uzf?zlEDL3=K>C^o-idkc=ev)lWW-*aRsYoe>b=Zj@QYro`sS`4Ve~yRoN4r^ zf=EeSqaVcvrdWPSfy_cEenPjHZV%AFoZRRw15C>p^r?HX=Y<6*AL<^{uB>CFv-Zy- zBqzZlAZ>uSuk2CGXUMe@5sgF$>~*(E41j%sJ?VS;J@ZSPvdA6^O*@FakF`xQoE$20 zMJEFE`~Qb^UKGuDL0ji{Wy=z9tAKC*#OL+7Zw&I5991kN=3}DX9i&Jz9a@;cCQA_@ z7hFE!Ap>0Vurn}RZ9d(cUl#CZHdz98ROqOsF+%V=%o21uqw%tesd92hIWfKLyBTQo z!nX&3A8e^Iqvs2Qp}Z!0s|cpA3x7Adh22HfoCqo$-?WI@PTEAdqPH1^c5!Ag?nFk< zZkKma{^rsnRF3(mIJXHJ{X4&*OC5aXMQ_^pDPAVeW9Smz6=fQqzAmd-#$#YY3pIvh zI|aG^PrH2MD{yWd;b|~SUaf02?`6wMNP$9vTb)khee@k8MDLgBj)F?_6%r23E37h< zO%eS1z~hUjnt1b1A%z3WCXC2uEv&OHaYf^2-WRhBzoCd|W_fhWT0F^b83#V;K~VNZ znLl1x=*0BOtL6TRSH6J4>g{yau~ng?&-xlc6OR2d#{6KoA^97^LL#+khj%2J-6ZMy zpt93M#JGtuL57IVi-pamL!moIJ+~Rh*?1~u2SFJ+^I_x|^`QIDPT4o$CW`c~z|`Oi z@$zZnx<_q?m9s-y^jw`e&gZw@?4AZ+&sN@M`x;@@a9sxPjC)RV6QsR;T?275>$ZCI zuvE5SAgST3_t60dMDG&#Qshwa)O$fL9Jsm& za6@a-RIdTB=X~X$7laN)_|(TZ=r|ow=(E?u=v|y< zLEUt_=~ghDlx}3FfrV@07|}ZeCXU16J^PUDboi#3$Bb2r1*i_5;K54ex=e>QoEo`E zt@9_PH)&-uE*|D&a|KJi=l+VceX@9eK4%KhaxOgyECxo4$$$|!YKN}>3w_j5-Q>wx z_2TGtMdk;6OZS~)A_0RoQ}G&gc$Ce@MhRciayB0gmdvJ+x8&HgGlBu z=*1A*LktWaPFV?2L{XLrq5o@i#jvYr`D6Vf?Ty1ri~yB4K@X%<0_PMg(aHq3Y^y(A z>5TKQUbBH5ihmkRuv!Qj&ZUu18N?li!(2`OWy!bIoM(YqBg$HX|9y7n7(HB6Q>y-%y*j`5_83B}~zF`PetNmX&F&NpTVbPd`N%25D}MVBdlU(A2Fq(qoR{ zj9sd0+_p__7;UK2ExKOh=XWH+;D$RJw@nV6@9u}SN@w%mlbz5A?~OvJNo^pNjbH=C zk&%)lsuFJm7~AE3AOb&#CoYlb%Zgx!|HA&uW0}AB%X;DbM%z(8r*TgIv1A(EY6#8a z4l1d!V3Dd)k(duhWj+T*LQ=+;F13)D&C83|)Th__5kJV@Rred?l;5pLXCZ(}{%xoyJ5#)JucNEAe zO4(KG6oW>G`#~L2wPB(T+ZkPEhqb0O53?V)szHSKt)^$FR^NzVzDHRXm8`tgE7-5Z z7y~`3Z}}N9x=P>Cz>L@fAXR(DDWld-q?<=UHfCDZHGActcR0hJf70jRSiac#>BdGV z#(HAYA9`MZOVS(BdP=h<%|JflXk2N?^Id&1BC&SRbbzZEKfT-$tK?A*D82)^eP9bv zj%=?Cr2hsK(&e{$&QU7SA=k$L5$U?YVNt)i)H54Pgt@qlQXZ~Y9Egp0(n^l510Y7o z8hh>oSZ7ot?ligHSix00)g`+Qj9ou(y^X8+HAENbFpQhF>2n-cW1|H>T+(8aE+ld8o#<~=? z7RZ*{^O&20g(^;0=l%9RiyaM)K$pk8 zjuG91&=ijqVs&^8ayIs;fsyPj=`x^mZLS{^Y2rBnmDW+utu*0Iw|Rg!X^6C-JvL!# zA7gUx8@#8LVfs{p>DW>`J&EBfZ#(kf(-$A;Z|W$Cy7r?CxQdrBv!{AbMXAWQLe#!@ z03JpC!zi24?fZ}azN-^37CwTpQ{%#^(3x2FL9Ux$iDhag^f~>#W}JqIsC2_~o_v+~ z4b<&R#93@pF}46YRzCLQ0oBEtw08`9vkS{ow60y2LafNEtmhA5dDso;LbIE;^YROd z?qsT)F^K*}BbhE7|I#?9wvoFsWdPp~PAC(d2z#{O-R9(mb!sFdRMlAHt=m0Pdc9?! zIg-1jXDgHJUdQ{U5|<_QNDx0rxp-oJpuAGCrhmuVdA;)pFCqF`ZO;X%3+g<&``eF+ zNG4_ML;kBSw1|^biJL8Iq;J<6cD+E~VQ82TS{tFKRF+AQ=5hsaiV22U3pr;e=oD?H z183mR7l1=0lTFpeaMQiYSyxJPXH1IUvN?PxIqs zfcP7nj&&FOqkquxT`-|@9EX>hjgjXgWhAt$-Qs0?bb!AD9|gT%h;CJZjPchZd>r0@ z0bGqQ=%9My=IlG*3_qU567Q(+9xwu(k$CsXTa!K0H3iq(u$xZB*}hLN=vTp^J{M*W z|6q{$_6QFvXIFSLA>%xUpn^ay|1Ll@4J@g#gcMb;-{;=%W4tlpl`i1>JupEjm}Flm zNkS5mh$OQbRY$gj^`;&3R$a+%l^fOD3YB#}r003x25-7MXmS3NFM`mH(PtKyZc^$+ zGx9)f_RJg-WAugqCmX@CSgQh$l2{9kN|Ycg45wS^7!-+;o0Wmdr%Q7+&+?!Bs6cQ0 zC|bhV$UV;rzh9K$b<>j)d2>=|*0X_~HG4$v{fw|HJ*_5h3F$Yn90KuOV* z5I!b0qV2`XC3WyV5UKY%%=`+p>oSg0=rcC*#r(6FmdM!w@3MSv z@xBx|^FkwGn9Rxt$wxWz#v~SuroIO)wGdj z>od$<$t(v!0G2$Y`L>S`Nkk~K?7}vkes_^!M_|5~9EJ-3NI8H)UuFAl84p`=pw4oN z?y?!r8+PMlvxMDE{0K5pu}V!8)0BDo%Lk>2uc<;j*6NrSV5Q2>tbZ2kH1W$){J=|; zZww;VV6npDta>UWV~I}j?`OQ4W>5My(sqUbwQySK~A9XO9w~v}8wOSG}?i z zCVAIp*%JYWm%lH79$cKS3fF2}u*kB)E7A3W@Y^3gnNgYTDn1)+`V+c-8!oFUN+>Jv z^IUYH!VFXPOy~w44=S>xISAj4xwIqxho|4;lcB!tTGQB+>mF4#PJp+Q2MN5$rYgVD zbl1N@70W|O^^vx?jTTLb8QN%zw2H11pIQc1J!iD;wd}1SO2XI$iTQuw4JJreoWiMu zXU-~W1Q#TZEzwlA`_MqFwsO@pq|2}SbUtJcg4vL?yTAFQyyj_Ho^1gZ*(A*^?mLTW z$a@7l0M`C{a-)-_2TcQsjTGl$0LD5aCXQM_CTnL5R=?X4HvFOv4LInt=hb8>t;@du zoRcV80dk84)ZpvAmN%tL>#NCQyzP!&)*fjsAGb325lA#uD%_$*K=h`g2ajU}Jqc`{8gygp#V zh|?SZ4}?=Sj;-6VhMx{P1h}%kW3kFb7wxM=vO)l%>Nl7bI#={)nVx4%g%Kg2WsJ5% z0f7YOjUt;!Dw(WOnC8zJ4i35->|P3Q^s(!9?4gG=6U=gd%Hb4+>+_jJcPnjkL4Z~l zG&-HE*bLo<;-ykR&PS=aP--!X30R@sYcJ9MXMO)#Y z(aL&nFid-kAE*_oRk&ouhAWc$hiF06$R?b}J{kEOuf`oGv}~Xeh;FsYnio-W$l_3k z7{5oOu_19Ts7jk+WOTaT^h@jCAN8#W(m}EDa_mH(jRVN+VqR zws;O#70Uv_ge?u@#48Ho%R7_wNdw!rX;^0knLis>-7JWj#xH7Xgq;oZfR7J679~^U z>Za?riziOJnCi9iB1Ust{h_9W`DihNmO#0OWSN;`wdO^(dFBaP$hAn|8_*kJ)L~jk zpoSNU|Gh5d)y(I%gfYe@NE05c<-N`~>`O?`^us$W&fn%K7jrd(1LQfUIyvd%T14F1QtN<&S}k_TRx# z59Tg-8O>X=wash)49LpBJy1eu39J?QjR8qPHDulsfBxdMP9?*EU~^fGe)l;?v*Iez zI`tuC?74d4*&IwdS$=phN=jh`j0bQ*HRqF(=|&GlNj#jm3BL(F8wWNfZwj2*2dWGg zyR0UTFN}r4@ZS=~o@tE@4xGB$pmCi!)~_uQ@k`ougVZ9dmD!8A56;lZO&~7W!eR}g zekn;tt!+@^=+h51`4i^IySkk1nebezc+TJxqE%F_srD_wxWq<)BV@8^4ONA{X1NL?_Yjk@&*! zXQ?0taW}5;Fka3`aV*wtz9hT!%JR7iOIZJPcBlz-oHE~r7Aa-(?xeFl6wear)YCSi ztxo&foajr_La2i4gxu9Nb6;fc(*HR_6MT#2dB%JYCk3up!x-t~q=6lGrVd(v)1spt z^Ftj5rMZUU1$@jKg(X_$6)(+gpnpZ>2R!<+mVHS=1)~dvF&ObY%Z8wEaD%XC$1Dxz z@~(gjc1`0~l(n35+I@Oq+6Xu_bYo{o%t(w+&`TL3**lL2)waChO-7gSawTQr(1Sj! zcSt7fnYbbG6Na@gzL*+ieZ5Chx3(2!zOsu-!(0#o^<5T+E}%T(Ik1s=NMp-^x$|k6 zyNac(;Tr+t-P^(KCp>WvVfM~7)z%tywO^vMG;$^?_OcE0g8iL&QfW10uyWBQgZKF#y5 zLB#=Vj7|6uzQ1YwAyIrZbd?T&;i)Y7HXukHwmD_6;l`&X4vh`$kJNz8pF-Dy4f_L) zmJmN@cp}xx0T>M~L=Ix@($Z-%VSy5_DnYZNgyS1+xzuD}f7fceg}5Q|IOZ4-Y7cqY z1U|g*XOonaOEX+WW#ybR-w-+{g|2I461#YWr6 zGiIMFdA7!-ce98cUVyLq>J@{R`Sc!9tQflc)MyyAo@IZ&NC$aN+iKAg%*+G;)%jZr0$x ze5LY0@#wdjYl_GRm$ZpPRzA%?FYqG!e=d?3(TmFPe=Z_@qd*~oiZdNUY(h#ZY#-5~ zzeEoRh%jX>MFK#zlu}X=6c|Y@Av+h{OZ3zlc@u8q9-3C$QGyB>5$n%zC&fNK;8*-* z*$7z9-Z6NW)FU(|R6hc|9OIk*y%%H;B$%HW78E{%@bye|5@KkQ@l^J5i>C#-t7JmBWP$grO=-;RB03=t>0)zDYS zZcN_){JX;}nBJJFOmrvD5oUBh!+=lNN4X{C_f1A-rM>jWoC%YRMVy2WY{kegxd6TP zO)KJyocC2yq|Er64n(-M+40xHsqwPw%QFP~MT_Bno(U@Gp?V_rP4U7Ni5@;ZUb7t7M{H%&iO>!q3_Ri}H3c-HHo+gS*=FKP-WF3$we*r!7Q;B=> z-I9DVw76{|6Qn2$S?BT-xJ9(7{8UukeINbg@!gI#B_faD-wU&PHf%HZ1m?tWNy~FP zJ|ds#gWzAtRn}}T&K67#5n3U=4c`?GrgIuJuixF7ui{3Ne$V=8jF+FA(X#&=MmD>H z6)O@G)2g|Wn5A62x4rifxS~n@RRYHB{NdttL*-JrooEEp&*x5xwJn%;j+MCHg!3sN zVp<(T!^3P*VSEOB@~m%TTX$oorEX!TxAR$=1@grh&ixzEP|?9{fR2JD&`QJ_!yi)U zh=}k8rw!Q`oO#~t#}aDrjw)-HiD!rhdD#klaQ3qy*KkA+iC4PI1j z^UgNp42bu(VurCa_4~!Q>3-TSr%W_i2}hUX^HvN%kBIu-k%^P2#y9t2r|rW;ZCa0> zM#Q6j6aUFgA(8rY-w|ho&M!`IlBV$pkAKd%P}>Gp2o-a z{HuH%eyuuv>CX_6Z-*m&T$!u*ml)|K&b>t9aE_D#6XVe1Y!s$=`TI>aANm`95*RT= zJ2%uv40K6kDwIoa3Oy9R5!mTsjxEmYO~l15%=kPP;3j7uH|H0Md`}t&wy-16y_QvrjQmhREGHEFz0?sP4`-q5L;2 zWv_QhHa1{S6s^g`yL(l}ss4EDD#2G(8%)q{od8H^1(8S4)Jgm1iz&423{;#1-GIvE zw12^^9JKe?w!P~r_()j5MS!!5`m6nY^P-6~QwTxG@S&(_Fh}q0Rq^^|0H{9rzTgX} zIbWl8oWLMm_j3B4ZZsO0MJqq@==~rqdovNbb*shGl#GARA1=xhcMMShl>E6Msb8G0 zH|_k0Ke0L}nE-1*llW5?IRy5+XCPB~2zKH+D<962PozIn-g~yqJ|ttrwU@}jl{N(k z1M7{1Kr&c7WRUbtsvL(ogph0@Fya+Ir{k^Az;PcCpLun1o=Fvxng1N)Ik*-arc;9~ z*=pn=VE{F*>^M-#!ovGVG__mr!2V6={KI{Ne?nPcSHhqbYDAKr`p`&sdeO$ha~0H1 z-EcdIW|VO_7b?|~qm_OR=SUy%p}_S}d9*wZ@v^CLfPJh@4ZF0}@iMSse2Q1Zh< ze2s$?)pAQvY&@#(eR=28BNP_=2~=Cs*(k6ja*mF6WN2rw$4~l)tP2fh1CswYsh9wS z(4{E;l>!AD>;6FSVt#M_$jcPBfvS(BoUmP%(fDl>RwcaAk0s6kD&=X!G|I}I=2hpl zE~t$h3sal*NwT~Ut zfM}+_7P>oKq~7&FY0*Q0^DoTGNUtTj@WSSK&Uw%Jqr{Gpl!MSVZ<0o3!Tou^tSFNY zG1PQp+V4nf-pxTX<9@d|Wuku3(%H#KKXz0;e;YTbB&YtT0;aeGw}DbD(CU{kZ6R>8-m?% zWcpG0C!=}VeXP>GvyxZ1y4O5iU7Dd{kqRZx;KIwWY`?Y4GagJYuERkmqD+iosIBa^ z#|GZAP5hpAK|B;|A$FX77~KpkQ`O^(jT<7&3{83@+o2IwS8&p}5d_yiT}E3D%N|mb zJ)lZXVj*D&GQSxKJid!pJIS&Mt-PIsXe978AfxIx>3XU*nT8|OH4pbuJ3y~_DG)gF z<;)qPMzrsTize|=+0iKjKek-kCwkIKd>@7htS5wUlXnl04yholdy7W&zG(Wks~D8( z$p``;`Sd%jRC0HnxXLDJ;wCb3@suzjCHa^B>95xddJ5sDIXbJ%@qXA@31rE4;|k6B zY$)=MdD~A-D^(P4b@{&gnwzQZ(-VCH@xav>t=Io z^~N2(K9@Nmm~i_xVlmQN6VnTRcm8UEJ=D}z-L_!d)1Bq)iM}(EFwSevAM2bw3MteR zFk>34*bu2GljxIri`z1Er*NkS3MeF`kNslg4 zWw7d@T(s6p$zswL>z|F|ab@R>k_E&m#Ju8G34^DOnID?Sh%(dNw;h!F{ho+^*1uxa ztd5FkKru^f$`_Oc&|Ra4QtwzGTwl*=^{R^Bz|U>!oV=u%MLVSyAUJ%u2TH7vc1GNw z7{-T9#a|wZY1TxxFq%1MIpL{kE+3`pDx)nR5?Sex6sXKkE4E>LaY%qRl8<}$&DdII zPG*D!NpKRvJr)$h{|I9uX&+Hq7|Wb?k z6tpFU3Y&>@mS%!QhYGp^+Qo9HIM2FqKM-k0BP-hTviq;W2{bpD`=LqM#ssiz4_NOz@?dk<>_+jAVnu37>|uQ- z6^Xkfzmrbl`={D>8}-I5#vNZNA(llIw0;!mAEk_Q;k}28Z+z%umJTQ99oPiRq(Jhu zB3)LmTi|?x#|*Tk*km=`EhCk~MC#*h?Oj)>XcVgyiTsUnZY zA^Bj?50JigHXJ$M(|v=Pp1`0Mnpj4x^|F&KtiE&ehnm#&K}skY>Y!pZ!~=i}S812t8}=z`o6`7APYL{5Zb z4_7w?BoA}WXunH7c_iz4I`taoMDd<4#W4Oy5$-I&#*6(Pa=0`0LCMMhU}&R zRVFvvz+L;VLd0~Y0mKX9clooe;f^ixk3XZ>#0>7e5%qd{`{BIr_e(>h1F`u;&7MB6 zutRpytYOd{D7gZXq$g4(AF>;MQOe9MbM^4QpW~X6N*3XKeh&{~EL+%|GfGn)JeK}n zrOUq?{bFelIqayR-PWOY0SHcSw?vi>rt`NL4}TRRX@(jg1?wV^Z#&9(kJCPHyDA}uI*+;}Ftk=U!yshm13)p5Up2h0X2DU)z!>w=Ky|0c>zGl9ovzXPQ9k0x~Nw>ZOZt(p0kWF2Z z`PO{NR^^&@vfH>>;%A`2oN&wiZmyuiHM$W%th6&i6&jdUg>TD@umK{!f)__gftELK zV77{9#0s#9r-Feio?!g?`KL*Ql0q^}Yiz}^9#JU)mrLMl?H|()zk08D^1Y2Ej2||4h;D3)ap~r+c$ZJQ>fh z;U1I=&1kFBF72gLdYvwpPxNNhxP?#?Rk`H;rj6%3OS!CZoc2uAh_fZ_-H(ek`MNxl z?w2R4rnF7ot5JP=TpupD;u=HBeTo7=`SQ}xj9*N zJKG-C&S5jBKRkT^S5)5=VG+cwG5*<=^n%=B{+V8agF)41;eM}Avqo*s^7UCvg6VQI z#${68sAFv>nTV3>$OQC|B8a{Yyacla)*8XmpCxRG{7`0jyfd%ABprs7F!@7^dZeDf z_ylg(B_o2(vb()bN-{D`#QuuTQS7vu)W3_!<#XoTs$*o13K2=O>@k`h#p zN(dGyf>hu}ce*iGmlRhuWoF{Vr>CT+C74B=xm5JV?pH`@mYH#4S?T{AUQO)b&-LsZ z3>-WBUMZy)`Ri~VnB_7PT`W({ID>i_j@-?X`$YG?P5}j};L0!HRztHfF^+^YMsSa=O@$ zvTsduo8j^7ub?qKcl9tg8(E$mpPSqstKH4yvQR2J*@aACB`19^s|0Ywfv+*u3z91@ zfI`~Ghk+t3F&N?5KlmnXKdH-y3n~_QQl*!a;Zk#yWJiV8v94C)a-P%H&{WaQOsKGGl!5WK0wPJ>;F{4?dFeZ2$)cDACn#Z@Q$_vMd$2&O8nBj<}d&uotEgfo}hOso>1 zDyk685$>*yFlsmy3D0Qg39;u*qyEMk6PJ@LN~MD;MYQ3##N2< zv4~mw5%0g*$=!4m^2l&4mu(B}mlDnv4%*1Crs9}*os^f$e%xP;}@Tlnr zW!h#124t1&B{%tAefeWn6S|CK`ot46%lk?5nH7V;PT@>xTQXK*$|9U;W_%^AuL<^( z4@1`NnDG}`spjkUlNN`|6}2#r&BIGRoI>uX$n;Deunp9#=l7C5xQ-Q~wg=Nl$m#8-8s12l(O!kf`%GyU`Cf-f`^Qcd=R2OJMk*nxvl4~; zx6Mk(tb;rdDys1sIYnqwEQtNpL}N^P_&d%U<9Z2km3xP~3xIMafr8O&?2Bz4G`&{3h#r0#3;)-?RuA zb0(O#KD}T*nKM~Xe`Wu|EMCK=_FWkwr4Qwg{_U5DJGZsgP=XCMnKf6%-=_$I+COz2 zZ=frdnbqiNY6=j;P;Py_&*n4+6)gormKu>27};O>XS_9CRo}@LYM8+uH|-I{<5<0A zkE5NXVixH<1pmKZ6n?C;cys=J6rAQ~!&+w~VTCd)vMhmk5ZIk|Hc%0ZK?IAl==Kq@;A0bO?*?mXeSZ>Fz}% zAtFeJfHX)g`kl-D+|TprdGCAwzq}vbvB%hh;og^vbzO7LYtG|5&fh^ZC6Ve9=K0eH zCB+$D(*>d^63;IQahKgY9&=I54?)g502m&o^*;Fy1`9|PD!IZ@Zq=oI=yUJ>8d zLwi77xNlCsFjJMTWt_yi_yY*^ZRD+_U5`|L@@+jm8QDiR*i6tK=^ zaxV(20(n<2P8OJGI7fkPddg$Jpfteb%iCR@5*~%)Z8qbcBP|>Qw43NXdeKkjp`UEu zYA6$SQRbh++8%4Im}^}kSd>-_czxPlx_%NzsZ&ttY|Fd-f6f0)6$pY>8FpZuSh5=B zTzfTelp8J3I7U=s`nQVLhJnXHy=x?4`jh(RWjO&2O|jBO%$Ur+80cx^5VN_oE7K&-JUQ(b9oD`CJ>Kd0*>y^H%c?EA->R!ZhtC?7|-@&w_UF z++eQoHu+dX zb(uXCLygK%w6KCHzaCRy%>N81WET^pn`y)4`%O0>2CJTFYw}+HZ!Kf)mQfM~bV=ab z2+tM1N7^+uS`@nafK>KN(k+VMm5NR*M&z(EkD^C7~D%% zOOGA17&wB7t4+ud;gIEngWuJhRy8ew3o^?OI-c$3BCD#?r-(l~F_k;bMp5}N6FmQ7 zO~1QjdSTMW$5FZIJCD20(WAsteS_+2xk<;nvRh57c;@Afa64%hF_EF?>>3)CmcNA; zn~AB~HD|8=?vge34eyg2+Ai(aN^H9G^EA6c)*bJ>M(PW;<4u^RA=*9ll;}(obYI>G z|0G)ApMHp)JbXZ)M62~~h^_YsL$`LIFN)WPxvAiXMfW@G_c9`|b;+3!A*S_yH(u=R z;0wc)d=Or~fLkSo*AS+yLN-V}-+t4ME6ja|KEjZ@Ep-a-SKMeY9KJxd!zV@!S8NdQ zB0kV7MsMTo<`=L!*OVrbr20|lM|%O%*mYfN z=Lmh^9lO>>#%En3E}{t7J&zb?YEI@})H z0%IaOm(>pPinGbC?6132BD+|q64VG!<&9Y3&<5Y-hx5wdX@DZ`&UOA0EuW$-}^2Fd(QW|Yq3Qp`f| zUMuc!<--_Hdn^h2)<*-3yK&Zm$8_>(zmiuu4qlyLG7b}pPNL#aC)y?j@kt{7ssrll zbVRSG4hR(NCNFIFqE{qBvmV@6IeYY*WCi{CL;qrsK-3KIMa@sY`G~#<>yr}hBY8YH zTL1B(O%~H&8be%#^2B=~n&D`h_yybtYz*ouE-+Ij*|itFqV%Pt3S`4LIM62{&#H!9 zc(yQXH8!dhZ6yB@R0z@3@5XK4S;Bep;HN!9R1awnf5Z?USsER~AC4msFbACWw{P>% zZt#^`!e>i6?{_j>zMN*n_%$>=HlMSSg`i|m`*s$++#3xw6VUk^T8{ zwfT49Co~`TQpmghMS6^;S;xVA&Yi$|IR4Kbd`a@ynO2#yh{B*+G7*UJ;NJVKu07hv z0ncjEvLwvO+EzbhzAMJZmM{t26Nw|6DYVR3)_1u7v@K|y zl>bH1-$FgovfAXgO*Fey2LTX1j@=R+6B?xmNqaD{oMa?u-!Rg z+8en~AV}dm9q08EDzJE2+TCqI}e^2da zB5=X=Siq}Z(UzKs4sgTd{!3v;o}zC15Cm;uuxgZk=> z+}RrS=SuH|2yZF|T>m5q&-#rtb@F_YKfa~Tq~w`p+w zK>wr%fm?!jG2^hipKqL$7s)fCbP~xOESQ<&5>=-yvF!vs6DMbxd~}1+I=3JGx9<Z!@q{$K~-*AGG=#jhL#>9uOA|DUg>HFSW&ciJ@`k z8X6(f)m$aXQ;R=X%oIF*R7#>3;#mZiMJS_6?A_aIV@kA25eK;mE>S|;2>hVG{e=IP zi*qgL=@y+P^yvNVqkzZ)`}-RZSxnby>DS%!`C6v;Id?^vZJ3z4J0Ii`A?U=J9@s%+ z9#LT(-f?7bnICFt=Ga;HhG5`9_uryDt4q92IC=uVY-n?Clqs1Qc%DzwcrUPH#B>1@ z4=E`m^oKILIe8F=DQS+yj^!9!B;c|LYHY#X@Q93tPvP{S74(47}8rp5+#3 ztL}VJJFe;PVYnfp-?xtUOtT;uvrs71MCtiuW?>z?E2v=%bM5Loaw($FQVn%*g>%H@nV>Qj`uz)Dk?fO z=w7$xv?ogB;%8-L-E~z;`|rNV-_P{-gnXW~YHn%iw&n~U-#FSDdzA8FWPNPI*+=g* z@Rt7!1Ol^j8rbDtq0#(??`DAr)FoyB>&^|JNJ!zrRufZ2$V2qG|4V+E=sc znH_f@pW=S3}?5DEn$p~_#HjSG8K4t1uAv2x=ybu%`c0d>Eatt9{3C-Gk&AVUo9 zLIxl8d)T3{u27-}?7sCH!2=l|LFLR}{0 zkA@a^cbP%{fBf72|NTP@ZruWIoTk0&#{Zopn{lrmCO#3YBXm~{8zP=SYp%>5$AS6> zoS7;~!A*7X2xvgEvB=ee98vU+XCjkT#4W0G+VnpZll)}IOxHVZSo zEzIcJLp0}{=aTzH_HAe7xyHA zuU@gbH%9v}M(cmwUy@s3LH>rHBtO#b%|Ov>;U81Oi^iklMtlZ&k} zpT>~uTvO=%C(r6O9+;@N3U^ihlZ*d9e$Z@-zT@&-gdvnZ?rZ=9y9{i8l)p*Ak)FI8 zbRDXYyH367mSHS=F>2i4^zTPgyE_#a#(iy&<-Qt#9$4(`yR5{MQOM-ejy1vQ8Tr@q zf>o3N-0ELhE;BZD%s_u@rlLF+44E-Yw+RbWfcq`X0mr9X-n0mRLQX*9q??~*ohb6rKEE1UooORUC%slfCuXym@K@vHcuzJh?sCL+@F=BKo8v@ARzTQ zgZD`FO4Sg%?1tO(1R+v?R3?2rxIX+Ig?qL;U?2ppu)XOFGU_JOWYRyWOir{>vHE#+ zx?nKtJTBk82M{jHH`jYNVUw*VHQhiLGhnLuu991O0|+g8vmoL(H%XH@#pQ?g{elKN zb6*gU8*Xs9Ox&G5e^daHC_Lnw)UOo4To78oxCAYZ*dwjL=g-gQ%d>+7JYmLNK;hWO zUeoSBbOSV0mLw42sgQo&0&a{y8%=xFEDfKxVr;SD#EdJK%Sz6|&^M7 zfv(4D;{9bKmq_==Hb{xbHfnD1vB%SAq%@3Irw>|54y=~_jzSI|ywdavcKuP!zeTH| z!0U~o>>Tw1W@pFZ;5TC5uE+DEW67qEYx7&WZdQ1bbshVS0EU6YCH2bk@bF39A;%w= z^m)k^q&R4Nv?l3A0?mk!1l4m;GqtuZ03SiLo~6%&`**Lx^Yj^*XxRxuEH)agUAEIM z(+6ASZ35NW)%TE$_nE#3RRB;(NG;1kT^0!QsG!{kNify*U?6o$Hv{5K$+4S^Tnq{G z$!E8g68Ez~1O3H7?+sZ4tm_-lqDk7CPH}yKpchc~0z;K0Nze7?>$H@Or8GWC%%R^9 z{rzo&8~()M6@yFr&VjqCB64Ph88<7+;muwC!E&_;+ZZp7?Q~Vsk z_1OLWXeiGXd?y8?(ns|NCi zsVjf@S^?hRKDu-6Qm!kz*qdHTUoSj~O3H}alf}ZC=p3!bWp;$vDeET#nyr=G08k_P zQRP~1Vja~#-;KI8`^SKCqio%kMS513Gr&G81de9 z8iyPBo;EG~eQo&{!s@_`WJ2O~5zso?eehUI?FQf>8E-N~+n(1tZ#b~bJQY+IL~TcZ z-}hJa`4cakqUmN1`{whDhO(&X-9bgPHxtIjw2hDs^5=+8t#cxsx44mxB&E0W0l3kT z)9WDMRkiuFWsi9ctWC-b-FM;8!d4M4YzO+#$64p!KP^wz7|fiJI?y9uESwFy$#}C( zQwOD9Y*#53X|*PwlvW4aOQ&Kpr^3ESbS;PsR zkFBk*cdWbmNHqG4UlbCz{RWgDW(A&&yrUPLo>LSc2w6CRV{`B}OZz?0$Y~W=9xa90 zTJ;uRbSUON_E2Q!)9SBtu!vt5`&AIws4v>(9Q*U2 z42K@lZa}i0K7fK=DjA2BNUq?Z3@tdy-HN+%1m1$?QqGzBJt>c*TrV5+gwa-_z?P8e z1~l#0GI5WMsFIx=d~>>4<7y+Vd`ucdZCpQq5hKs?KyTt@y%#rxq>nm1sF>k!nAffX zlz`rQhRfNZ*_b5Lh{2#b>Ob?hQ6ak90lt?lrJ$2BCl2ovhVz>I#f*K|1JQOiTg4hm zw4RIz8D6y`ZJU}2tV;8b%1a^HEDN6kr@hLB*j4_hg9piVJClpg4e$H@kX%{3&>!IH z#m&&IUD9pxKYASAZa?>==dxK)gEWqYWJ5%KKyB|;RJzZ&{11}hSN;J)f1 zmE=^4EtSoWezf_)UezG#nzEA(-9frFr;{7auMPC`b?il~2af0vA|(wyzAH;cVlW!H zh!r!S5<6`aZ@|Cr-&x&XM8c$0XhXw1;KYu(=hSGz`qC02&6t8g_u*^!=#^P4J_q47 z4B7GO!wr&ycjH>lr>Y$Q;>&u9|C)=C2H)78QWubGmI#&UF*RAh%C%yl)*2@!?hAdL zDpZ*)AOnWO!s*Ki^0SOwcQ#@v_Iei(pd*HzSdP7T__!;@T}UqImrqV~m$z8=C@5!g zUp|LJjeNry%m8|BhVgW6grybCy^vNC>J)$&gQ>x(F-Fdu3=Y%3E#$out|L^r@W|Yi zMI%UYz*J&;k0rf>o#n4ZNbL_`cPxD6(`8vcy@v^w18`_jpNg3pK1>+)n#Ef@GYd`% zdg(FG!`0I1Y8ND5K%=RyHd!?u|fMDiRn-2yr zn2A6DO@F?`$fk-G*mNex+EQySZ@dbi?P7&Pp@z{gG9S%pwPN0Q>&USv)hqTI0KoxR zqcBDhPdWBZkQYQef7k^4ARbi8e$ISeuTAh2c6z60LG_bm$pP6rkJW2)&gW5)3*9YgX=wck!C;K|3qO4O&FuKw z=fmyXtYGD(l(1N}=wGCYIRw*&J=;=^TeU(1+1d@hm_tJlDJ1RZz6FM==|$sO4l|2T z=|RNQM@pGW4&&d*Qj6IybFDpHZs0)ERlfL2Dw`K<*P2!)4yXUub`4q?QxsMS7*zMD zbj~yOXh(x?2@gC};HHwx{R(C3`0*A_hiM;oces5A>~qQ6EH7jxUR3%!ET`=9Q>vLk zj{(;V_X2r7*+x;2W6{+0OSW>YA3=vVv>@tyuJv9fnLC1n+K9&+&c}8)?snh}IK$q7 zwsRzyKa*HvCUhi?F=8#v5Q;O089fnug-x)U>gQ}&B-c@GqDCrel8rj?lHfv&-qWG| z&N>@17fH-;*T)v2_G<);qkz8X0X7%Lsh7vADFgTp6cR4<1t|&-D3cF`4TzPZLi@Fu;{w6Hm3E+vh?O7FzQV&PA$v%(4CN@|Rbe{u}vpI@)SKc1kVFb{mQ6VGS zX~;xO8PW)vvNws*W`)T@pIk4DcwhXhfl_NIqdt+_QRxD$O!*T z5#`!sUv-oVJ)JBQs{EXVe?w@R3Gee`KXvm-h*>wpXII$uIXrgXI5PU2<1ZE5jWmKjMs!Q2h+@84NE zxx_BvxT(Fb3}q)2%w`wC_Fz9A-2!1ce^Of)NgiM^dT_e(C;C3U{B!${DP4ItNp7@D~=8SOmO+o9w5&0tyUI4uQ@rnYk0nfm9)<3> z@cP5GY%C{roLYXhl^O+YPH1s1vA+A8De-^S_juI$?o#?)_#2VB6#fAxz7#bRj5We9 zgdwv+gkJsG!2`UWU*opD3$*Oy5`@wOJW|`)akT5IW=x*~#(}4X;xez&MVAijp$$hN zMicG2^%EQQ@Owt$^Xvv6EiXW5+O*TMHsUKD3#+NVZFN@l)!8-!#xp6R0$-Uw6h2?w z+Aii;X`SoP`Wt;De#s;bsznfuqz+_)=Vp&gsnxl9J7ToZl#ZlsyU)RiCkl(D<3i*i zACk1bAvY;_ha14d5Gjdo*~%P>7AMEqhyyK;PO|hSRwo^zj(C`KeHn_{03|5{%>e0NfmeZd(;vXN7EDyuz z6Vwl)F>w%i2ZS%c7rexSCXvQxOErd7OjUB|0>_o81M_l3MU3RncjJTXK)4)aS@n=v z^aZE->*#K!`PBgCO8k2kU~s-;h7Vv+uq{dvmC*U?V<50NF2Aaoa?)g)*gq(F42k+k zS*Z>Jes8{hVF@sVq%@^|gA1#pWm-#`$92?vua7#?!h)&tv;uhwT^#CrV*i0;-;zEi zB7ht+2R9vvC{~nHRHyy;Y^KK)rEH)}``uo7>c?~bJ6Dpn)84uO-(>M&ZWvDBfeVvw z=8xmmWhmF}j;Cv~(UJkABc@Em*3^!9GBkHOOeg5auYT?2vc~(6PNtscoHcH&Kh>Q9 zTYr)Kr6b;X1D~_iH}(Pkric*rBpe}{ygUMxhyEtKl<6{m)I>W;)Sg#OWwKr2PKcI7 z3@F14ln-JAgi2=uSw})+1ryQ!01%%HAYvq%vo>RF;v!XAV zrH&;7dN^Zvaoc#M&EJujodq%Sx}S4PRW~X0x+JRCi9pcbMiCGzzOCE-to2Ao@VK0Z zVGJD9?K9)$B=g7PBbV;1Vp5y-T7&&U&xlZva3@B9=ycrdw_1KeuajJUD*{IP&%+fn z>TPtP?`!!8_2ak$45X>u)m{4P_bLZwUL_wnsi$Xs?z*tOjQIQdSAH~8T_O?uJK5Yf`)&()+L^{^^W-1ORbqPs( z<`Q5p5^`Cxen*)v0Ub3lnv=AI!$u%~)_{Dpe`~EH`c8>-whr1xwv&Js`KkCbABL1S zXN#t~^uDxyeVnuDrWdQ8Xoi<%t{mMbgqphej-7 zddqM71KmMJpS*SAPbRtSwCZ(W3Jal)-|3HJ#s#0!F*Dl@-TBQEZs*5U< zFW?wo@jo2$ca&BO5;M{hFaZE3P1nX!*SAwkP3bs8^6TfF1GwJ5MN!J+quN!D!1|qY z^NMMGby|Liotlp&S2w4$69#q`fOFtgi#AJj7to>g+j?(baz)%ZhwzRISvSrnJW~z$ z60{I3>_Ectk~Zm-?b||2DN6_E+>!FPtK)Cr>^Rch{#kz2M2Wfq85oN+``fu+hwe4=*o3EqpM+cZ8?U~~Os%#$JFFRR-8>>( z;b_#Wgap2;W$6-{e@I-CbAuYf|{U7AR9 zs@HiU?h)fXF7C7*Z%fK*u|$TQU+v)sf0Jwghj9^VZ!)V~G#<9sVtHEIH{u#<>7VR+ zu-VffW*UFT_+alnWj<>!X!ppK&y$mb<+Jg2#O*lmlb`g3%SGS(jODw!{04!>MpzoTAo#N7UHITSnRDZQi3fq)ikTwf$K+*9NL6*f8NS zb}!u{D73N!gB~S{2$T023f$(q7N=ng*)}O2iE(HF)P&yhPDbi`=+Z@hL}{~u9L{kZPr1uFf(*P);^+@^yhp7zEYQ_nNEO@1p%&z zh?p07R)=Rn4^!3Ck8-43_)?q)72}6MnbDCbR*s=<1xSlLgOF3?jV}E}9eA52E0sOl znOQ^Y_c>MGoR`d1;L}lBZ;dk2Wsn^kp{k!&gOYxFPG1;a&sy{Jy9Iq1)x%6}G}hq^ zr?4J=3^yNNhE$b~P|pvRIgn;dEzMA(Wr0Qy`H}&Y6iYSOn!cl_iIRrPA?`S1pQ$bL zC?$-((g=YD!8h+WUP+OzbiU8s^aRJ^3>k-Nz19=co$Fw`nGa@(> zEq}^&MSe+~TDPwH`op4X-@A!Za89_+N1|;n_ei(+^5at<>*QY^pyxTrQo`d{^(0#H zfuzxQ29^l>)(n6>*HTN@5arII_d$z(g2cHt8Lo5SF6V|b8keWCq2|= z8>rjepruH)Hf<24Xv2gk(S1X)B4D1eu(-z#^}qz#)=n&O^x-vc4iy`fk;Efdr(O4Y ze!1AHwpc1cRhz!YOhu|{p zS;Qi>v&fIDJpN}(+l*%@BT$gu<8InGU+^Q6C4>ZfAs#D?+$oec2_w`*PH~7Gy{(8S z5q+PV1raY-0l#${pO)xKj$n+Hz1qXH(P?m zoYd7&p*m1n2==2N*XBUV3dL<$_E|pXC~!Oj#5Qf#7It+C3^OBnNZj1{6b9-3A+U`L zh8WV!bls{oHBuU8#0j8*QrSyoL7$KtX)}~=yHVz+U%(HUt9XE5BP#v-9VjG-_IN$g zeSt6KIKnQ^FT7uAvH1$!n`l5(DOtDUFqKB&uU4}KIhGM`2|wPK*ElJ0aL=mk zsf|XL1*a&u1*U}LnAx0-o+4`M2x5qh{Q#2W!|74m)BZ(b)q8au1!gNhKlDZ(+JF~F zDm#wDW07H!y1~;xK_Qj9MCN1zlc8Bn_$o)6=Bz?}YQ9Jg_;V6SKC=XQXj@DVg{Q?x zeJx#uBRYriO?uES!_9mDNx++7MhF|HKlb=DaLo4dnE>Tv#7%(U0F?7$wI<5N@^D2U z4M}>{8ok_5eHdPdhq#iu>wzBQw20;@D9`be1*pZrSs}&9!y>j-P9qgWJ2_2D z^0#B27ES2S&=}MyGY|^4`o%oaw$sPO-eUOtBUWbb-zL9RuE9mQ2LkE80NO_|ggmZd zpI_DZBrZiu{zI(T=kLWq!J2w5>V+p?8euY|ENm#F#`kt%9x z5y7J|menv!d=^9q-3BLFfC#~`)b8>J9_+O2x+NqC=UYlBi=%QQX zy)LM;>@UbB_B$#(!&~D1CJG~p+PFp`0Uw;Q}24*~j33yS|P-c$G!v;Ado5lDnP0lq0qa5lyp{9f~S)3)b+E##9n zC_PsU4SQ{D{!lKDnq}mxZKV0nSL+`hqF))w5#1)oY=>2|*~8gVl5mKUo_ta;I6x_8 znCM+ZQYT#_?Ke~wv@fSQ(8UI)KNRecUf2;ZnvZg51X;TYIul4a3hs;-e~%|~nBXS# zY;2yp;IR>y;X2?r_oj59$hf2~R^ot0q8f<10jH64zW0VCrdP4YQ847TS#g7=dX`uM zmQ#r+UCHS&cv)zo$xoLS8=rtCfKq9}_;t&sRL0_{Jx6e+b%*RnrsM>t z;CkZ{H0O?$L)XNB0INqvY?Q%}^G0{r-BVjPPT;nYW_QJ?u=!eMVOG~3=w*x{7dn)M= zq;#;?^zVuCW+bB)1dwB{WIiej z{6qYx6=0m^HhOvyGGITwLxJ5=I&7o6@nE&kq|ccHvWHQ-WK59PxscCJLyCZ0WA2O7 z^ocJ&@Mfya#%C`8r z+VQPv##4mIg+&7rQ2V|8tedI{kcz^4y<|Z1B72XxOvm5kA!T}mJtSYQCJ-Z}=h9GtQ@j^{A{1tdZKFif{aC{X#){bd`o23= zha(ybttW@UyWoP^x`Z+BcSx4-W5ve-EJ1dn>O$!BfS-=t&%{ym(q;tg&LR^}Wug6h z3TU^2M+Ua>V8M%b7UPwtvg0Rg4%5$3Yy9T9xbpA51+u^~7#US*-hgsxQdD%@MIX!_ z3hiuzz9}5qpQHpuY9r7l#Rw8s0tXnxV4Ui-rcYy-XtvnF04ISo+Ef(}7uh!H*YfYY z=%hUl2ljbs_spcSb<}_N!V@x-5)9TNuj*s{9Cg4a%+yc9hgn&H<@WA}x=5GSSZvV4 z%OC7CJt34?3Oi4_Lo8m(#(Ad%q#?h7%wo#os&Y^rxXUqWh}0_;ytzEQNBLR7pmei2 zG#Y*0|nIZxSCu^nisWg|#PCc9F|r8`VD=weLx3BH59&v$4otYfpx&f5yk zk-XDoN|2)c*T%t;B`CLHi2E(rySZ^veeKBrSTaY=}fzi?GC`?<*_ z0kXu?KzZU3y?=*Z0paYMkS7J0Lh%lrM-tc=D{!Xz#Q+;$jG!cLRqR3s9G9|m;p4^& zXMWUK&;{c==~0+^Q%Zmz1BR)@@l&2uRLPrUh*78auaMM}cI(cf8kUh?y6_uxu0evv zM*fk;R`QP`1PHwQD5u1@+)6;qZ3kizW`aHpx}KtI51*pQq z%?zeM6f6*M<)hRm{UVts*`BnS4#CWg75>`(VDjE+O3l~P9Z1~OZtnM%=w*-I$;Q){ zlL`wVVba(?IQkM^gNvs$ms^ERR5ffEcLhH`7VOmwSf3Td;Hp<#F1_tV>3dpl+oTyV zFdGe-!S7nraBXpTvipGH-e~kJO7F^I^;s`LRl?ukx8r6r)BgOx*(T#Wv2~{8LUuqa zF~c>f>!iM0sbIjSK?u`N(ln0Xr90!+Fxzk*;t>h+xgb5@3w^+z&%VrEAXcC^^iU)M``pg-rT%4v&%C6| z74>IgqM?B91#w%Li#??rIO5BNlazQ=J_y{Dqt)*P%V&vFM~h_=mDfmhr%I#Z-MD8D zQ0v*sdm-&~br)Ng>ElY9EMO`btuwxi+OP}Fs@}3oLXESS_I=FJEaAd>pq@n|H4(GR zFUAm7K(47!rq`f*fmo+Dlf5a7!ru$K!jBUD_Rv8+lm0Chp5(F<0S&D~a?tTJ)q)`3 zBU7ml1tS4jZkGqWldnuHR%Op!-zkT z>Os!J*OSk@PWVuyw6x`RVQv&zV)xfSFRIdHgaGW|QTPnmU(gMYDEX&LyWq%vvz#>S zs5_@NiEP|Mul~42rc@5~mrU1g7`~mj)_smLemwW*3DoCvBZUpzW!+#~J}NP3q*56D zE8BCSA>Ycyy8&u?NSF*VTs<&$F4I%bAgc6DL2YTVj}SE6kh&cHq=+++ZB&+swOlPV zr%^t&+aiB_BiQ(TzH8@ut+FQuB*L_4!sz@M2>dU(W!WMIw#Yu@fOYR8yY>(Jf+yKL z`0LW98MAn~ROhqccqmbf(^Ir4dD`LI10T9Cv(kMLg()(Ha;XUZ^v50hYDls+ zU%=++Opt+OZDGkS*S5-l*W_<{e^hzs4K!+f1$gyT&xba95AetkdFTZRI*$FcjBYxg z%KN^%R2O-~i8qe_HiuegGY`yUInOElofaDv9*UkqE+&@&l@`g`pJq-XY&NyDOUZp> zYMGDkA9{D6I06(3&N$iHf~L0_UJA^=`ReOr`OMnsviZ@Lz$RdpO!gnM!K0!BQiu(r z2l$b#$Og52>Akzf7Ci{%%V>7NXS?Z|ala2j4h{#f1wY9g1uKN+I>*zI||4Oe_tkhS!)e&IHCu%GjZlgF_z?zWX)@nj#$XQ3??QKaJg&y}`hzH&R$ul_S*e9T-kW9BcK1 z_rVB@RMImJFQrTy#FxSX1?vX}fL(S?*PU);{d@szSx6^o&N8>4I5Ols)1a{P(Ck9` z;o^#0tBj%9KMqD`id)OVTvCru4jLPP<^@T3Yx1`G1!gskkY8Nsy@Su%ag00DPy=IibZmfQ6c`TA)0>& zdwBn=24AwQNetRlfr25U?gdmT1N=&h9;cMu-Se9?eWDE!#7d3wT=0?Lvpo-cy0Y7}hjXV0`}dEU;*-+d6Vr-6 zDY~Kr_pbF)W83|4#}vf0YoAP{lfKw_v{yR5cDjigvc+4(i3fej%2y%9>liMMY!0Qt z!wV{<@C3Vs=#jB#a|0CpM#sTHFlt=@weWnWRr+C;Xfso!H&VCsQ`S?4ov!~KpKq=Z z1E^`Cv`-k`CwM^&p7a-F8nT!V*GLd1c*veZwG|T~vK`zi#^v*B=36q=XEBu*p_^)v zh)+GAI`t&6e8;KM6Zwa^+52pt`Hp>fL9<*=YZVRD!_txac07_61I~8JQ z%km&92pDlbR6y%@)_Z{mr%e_2LlU#epY!}5ffHBf3-|Vle|$DYY6%2asp&9gIMSuB zI&Ya$y8+0kuB~BNw5(yb?z}V6hbACm`*%QVGq&oo!$R)>*BQ^5Ie9g((Q?QMOWBtU z8MVh?iOPAZm~j)Dq%88)lx(WjG7qL<2W#vMuA*ekMk=s754X7I{#oUx-=>w^gY?H4 zFN<08)d&B$Tr#1`kEIwf%tOv&*z+7ORrP7VOdb}m3{uhM;fl4l=}|t#8CX}Riw1JO z(ob|MO*yQA5nh}hDrXKrIeK8A+8^upL?TfVozxML(Y$C(S`$0<*^lJI}h3V^25C?pQ!N!nlHcR1MRA-hItqv4n=YU zE$IuC4lQ)$7(saGR1!u5oS#G11OHZJnJ{L7wQ1>9uM>PT)>InPWjI{89_0vH@{pbp z9N{~Rm}WY@Up#$Uubk0JZF74>2|EYIQTA+mAQPz`nqd84uQ??3rai;gr|5?*YIGT zZk)?6FveKR4DWz-y8!YI6vm={7U?P0EloXo+5j;0;eh?Z`$%R<9jsLaENDlPB#f_t zkRpPOla+>id_cS`Z>mB20#@5E2XHwana;*)PbFpd3%FXI!MfQ$cMm;7G_YQpFUKL4 zpC*WP+cC&rZ<`C2c&8K}nhC-}@i}NI^vuZPVK0JG;q1mk;ZAF_s%+-1-l;+a9ke&_ z5{Iu1_^14fes?;Tfq}7Quj1g+-6yHNp7yV&^SIyqnzM!w1NY`N4xCd6t?aRXTrABE zx_jZ5ts|{w%A}eChRkV8!-870r}-h)^5dkf=1+F{<-v};nq92Zd%W04rSYBBD3jO7 zmFDDY^|4OQGU2tNGT~|E^cwQKqB5wK)~A=*+?yBm+3Fs2rQ^jPI1+6dYmyEh)~?%q zq@CJ!b(%!!o1-bWMg%kYnc3V|1zwMbE1*m%UGK^)K1q~KZn{VEb%6z~Re7NaybNO6 zwFv4#Wuu@rKxT<@@7^0wBKXnn;;vtdUM8!U`|>@gL&4cg$Fch4s}!u>C4Bj??oi@* z{$k-ECz0vJAEr!p(}BKIWI@>`&Qg;GtG$WGIbEiv6`qfY1LG1iYi=ZJpDCI2$?ok# zWTr2wT7RxSb6Qrq_vPzdV%<^K7F(}v4e_E_CyTp!vZph}43B`mwM_I3X|pU7SfcB7_R7dvyw<0 zzgsfn4`Ezr2zh-!Wm*)l300A4D|tOd(4vs9*ze%6TJ_H~!9F0y&b~c71y6I| z%Ze5)!S+WQ4zJh9yNqQ`Pzc!0rOm^>$eSB3ON6q&ti3>aD6WqSM=JDeXe}bnKko*( zH?s@URN7)H%esZI)B^Kt$X1c1=bs0>HRh!YFEr+htsf5O)BgT2oo2==c=R0*uwQsB zkiE#9Ch&>!qV#_1m8s!4-1p~5KEkb*;r*xZPaX^9oLL#QhaB%(qR)4Tmo)3J1R=G| zIU6tJkfQ_&QJ2rv1z1n-`k1f~`TBY#DfpJRkoW#or}9w6fenG>3!rW$xzqFM=%8h0 zW@Q~9Q9HOUdOzIn3}5i8Spomt(Vf=k`|$qQSVeX5jAGSmtJ?H0%`YF1e%C-7f3eVw zi?l35tRWqKC2A=$lDb*Sx^vega&y+aEIavbp5^WO%X7Y(AhPpXN1dh1ii|kJtUC-> zeLd3m>v6T&@1pM=)dkeJTh)+Q(Dj|j$5+YB?in6aX1#8~s+4Yp9edJ_G!F>86ozZvuz98eq43JR#w-mUTE zOa76q@^mGfx^04$D8XnFrW7JHh)>dB(D&=@1Dt4U35i|+=8wj5Pm9>77m-cSvEM^c zm%F_VzY-G@pDy_N78Z(%Q?m2C79M5Yp=MN?2)16*5PCV=W@4nO2ji;DD?$^SP}_AY zE_Ak<(+J;iz4^?7Ix>U_WfViLWzktiR0_n6d1MMC+F#$1x;tK};$n$vE5HFklV4_U z{+=Q3p~)4=bqihcRGM&2zs;Tg^nMQsS{6ROy<3V^z-3{MnqyGcLS3l^_&}iC3c-bR z8B{N&IE6{Sz%d6)^|dgTU>%`5<7<(lm?gQ1Yc0}ZLUP8}fh1Hb^`)IAA8KQ)vIhhw zd@6njS*75_7b~dUqZ{NcyF3E@68vg*DZrU8l6aJGXT-ahtVZm}e*^{3jJ3k#wL3H@ zb(TIZ0D!+r=dZTj)(reAOYkDMLLaG%r1LA*Z06hm^QM(GdeN8984$+RZWWbAy5+zR zIxFc_=1F<%XwYk%*wXW?45UGH$66YiZjJqL!a;d@3lZwI3KQ&~@e4pcj8n(Lohw!b zwPl^W37VA?wf2v%Z>b_w*$g9J{(u|2-@G%MtCnFDc6!tln^?(39^F=!v`C|dmM^m@ zYjFO}npqIFOC_PWUHnKj5+|4009aLLD9E#dv2L1eYk!*2u$9&;m;RU{6T|lDB9LXm zZ)5ljg4UEg-kVcR$f1h8v3mpa-**^Qmhxzuu@l{k9P3do-MJ<6>LC23`m)>otdDOK z2F_@6B^DH)j1P~%*9U5cltjCCe}^ka#tyJoD<64UuoMs1&HVLJFNBd=748o?|3>m$ z){;yJ6|v3}C1~y)7bCy{ECtK=FOVT=7YM>GW-=&UuI#W!s%h3LNi6LaG%E$GzX_CNql4nUr}u% z$W&e4nH~Y^iK99IgH?K#c^vw_`==2+mw;ZPWK`yJDqV2@_6N0yUKZMG*_R`L;>tzw zO%roS4b}d1wjOmm*A#INr zA3ompg#yLRsbO~!#S1r;oAl|U3Wn@>? z!kS|kb3X?sg9pKf{X^|R>31GdO;ktcs5u(s1Ic!Q0m7*`o^8kY(FEHQ?7%@ta>{OH zhRe#|P}pZA_U%(#SFivrI63C!QV4Zl_;SEeJ@ONrez_{ek2FB=TnoGXTb+nWN4Ae` z*}z6OfQCIHqG|qBzl|#}g>v2oSTA+?-q84=-xG^r731rF<Nzc;Mg1B1!gF(r%8QyAzT|wrrv_Gwme74uq($M~M`SYh|4oKZofz8#{ z=e)MMKcws1-?2&*hEo;(rn4vaj{4%TCFpbKTzl4Y_S95@8og65D!jXm6oP&E#1I~U zAFZ}SQ+l*;^AQ;kEq4&aGCrJAi^2BKt5XT$-cwTy4SDvP#Bj^R;@h1A=4IWYCE}QB zA;HLXCL1a?t!a*F(Fk*{9^9%bZcL*bLdae+FtcP!FbMwo=^vxO!ftC%9A+- z7l^ZX&q(F>a`msMFGDr*zb;Pp9MV|?p7;V%4yL z_suw^q0Z5LP)qUt0;H;4(jTiFSENyxjGqZ>R22uQ;MM38Qf?o^shw+JW*h)PL!w{%Mg64H_jkZx&-MR#|+^Y%H< zInVok+U~FC{kE@b3u|%TbB;OY82|roSQ*bW4=7*k{nj_n;0JgMreJ&@XQe?$(FJmP zOR*Ik32jje3i94Zk8c~ELT*J*d(n@=w@hkkrG1;V=X1EZ%{>!>JfFk0wu(p%?;1wm zRz59%KG)WOwU5V~RZER}q#=zPYRMjJ1$LP5Pe`0vkWodd*c6^?f}7Yve9bbZKzNkk znL-(Fifd#VK|UzTLWTkkh70MQ;B{v{#-M|=3^b8g|C z_yf1tQP3>4t#XSgLyUe8|4>uX9Yz}O1n5@=MURxG!qclsvu5FL%Hj`ia0{0wuFNlJ zoA*I;6WbH6+yNHQ00gwDonO# zvQZ=r&0r4HUnDN_;Yy@^qGWQn>*F-EO?HDPp2_m0d&zR)hgOZ71V8(pxblKwBe z`KROfV>WtH2OLS$3SD!~vKaElu!u~XI!7G_5!7Fp{g7%n(#N{eat_}kK&(xx@qBcR z1>~3&PfZzs0ix&Q!S@|ER-CUb4{1A-d>+E{N(;1c!2ocgc_|u2vXDzzk3A`BeB3b zBeRR-j4bGX%gNm-D#ASjf49=LoXE|T@SA9HW$dU$N(#30Sb$86I6dbBEcVU1I-1GV zThw;4U@8uhBeX@kUj-;3TYcR-P~Hd3eY=M5G--^Cn+p5pCPVt*8{nF1p@`yi* z$vH`yWDW;E9VjRcZc2N3&{cR5weBH|h~{;i{x$o~~>k`%a+RKk4GP>eo=K>TesidtRVDx-0dD zRPK4ra}sw*rxTo*>5>)7)1|csInn$;cY((aV{i1gYv31ELXFCw{r&*71b*``-+tca z{F>}>eP$fA`z}R2eHuXUgFB-cZbdGFLa1gB@R{vd8v4jpYA8X_{s8^Kh>@yqmEgQx01biDf5E7;- zCi9ghDd|BnJdj3#rSrEwo*3T~oFFBqOROF`J=|k@ZNQwUub99_t=RZ9yVZpltHhCri`^Bvp*~Xu9gJWFTe-&hl<&M_p|=%d*P&jhnGRidRsCorY(ibYN#o1 zN}9N)lRoJ_G%9}K2m9%lr`dn}YeVh;MulnXOcs=H{>dYN{wcyjFMZMf1<)ZOjiLZf**{*RVOU^S84{g zn%yMU%KW_m@t+^mR80$3482-FSe@Vi!9U>d$fjGOaO$CVrUq7{{`tp)TC@w;rg;}> z1K9uhwf=4hqGCuFyoE-#%At1m~;#(sYS9Jw-A56JQJvY zYnu9N!S(k6=Ztrss3@N;@s+&zeo_s-p}w;=)a-~XKm z&n6&YG|+Q98ifkZT4Zm2P$TAJgl>WQF)By`$dQgNHjJ?!_`3Ax+y};x2ETw5Gf3?5Z+2(yuSX&5~!rGV&&$>*& z&UH=jy0MoXAE=0;uFan&uhp!t;&0OXeb4osEl!TfYVA^kkY{3Hz zbAsE^=JEPGm)&EwCnSwGnZYcNuK^ms$F2UX_cevicSARnZ2*P^-SE%vHcLB#;_Dab z=C#3a(U+^;A!KYW!%SJZ4NaW=LTY**S4ZB=rSLP*b)c!JiWPk2CZ^i}_%)Tf zIjcO&5!*#mJZaV2Pfw-^s@akrl+JMf)Xo}NQ3xL{ap9e@Z2U!Vv0OC{DkK-*6i#TN z_J=jfjBxk?>WFRm{ub1E4|$CO4fzb-D6V5kUcC5q(tsPk?${o1JJ8&C=fhZCP&|NH zv_X891yIyiEd{f22_c<9o_ul1%6$lkFVXW=`O;xJxtf8-Gcf_AUeq_I@7XZ( z-bC5h1OsC7Yj)c7|rc;8(zO5R_|tYOntx1_zP?VqG0AS z*=vod{^=@+h2yEFdCc2swES-t&Z{~~FAd8SCrwO>_AA%@=rray9gD=stxHl8v3Vet zHVVO-2AxK9R0pBDYp}>3fIuoy3JS;Ho-##ILFk)UVW!#rxGJAM>nd*n+k30=P+&M0 zoeLA(e$)V1Erj7B*QcMq_ky6*_M1iXwHYI;!MbV=;H_Nt0HddI$io#NjGoma$afw= zdYFnSrFR(4`;^1vTt~hxd(NzYTC=-(17_^TrOLkx-}S^xRdyMa-Ba~G7UUT?w4eKK%0>O%3n7-N=QBKoHoA~M9Fc@nO4087 zm9vj8YaDda1ygq)7Z#~b0~;8XUaub5D!ii zzRe~VaSFx8W_dr$X~%5$`9_K1Gkf$E>|3@&*>Yl-74v?i4**`xsCMFq3r7Z)BqU_O z?*8#MBlGSGh>1y|1ypMw>I*vj%(8?3hh8J+tCt!Za33Iw(tIrGOhW!& zb`9jPME73Bep2zQ6indoS@&0$q2c$tEexI>kON8>6>oP^05>22yXoOcx}CbHx5>

#*M}g(r|d_2=^6zP|ss?8&~fHxa9jZ6YwuZ00K2tDrqMs+(1LhbbjK`bZQ$ zKic!TzakH*x#tLsA)E8_2c_C@6{8LfwaZUiY#&$Pm%f;mm1|D=X!kCPb_qX) z74YLkx#v|0rbd!d@tx9Fvr0aB)|1BC#*}Z5Z`B8=e4){ zzn~;fy^PXfhTvQVdj@{d-~PEzfCw}H95kI`iyuJU$*LFp$d#M;+xs}K8t7x3g-B)ey+gC~!pM zwu(1?JPm`ccT#0r3r&H9xCEY4_D!q| ze8^5|+mYMgt#Bd*b`J7Ezsq=7rcROfa*-KJxl|3tKQr_ceeL)LlY(emYZ?Wfe|9KQ zgk|mmxrkdMp9*r3DrMCewt<3$AN;n6PANz%;;`PMFb9`VU#Kz)!jNe8Z_F$5S+~2e zIHI`@62)Q7o$I5RKd!j-qqso#`O)c{ zza`GxIL#pofsqQ%iT?>|2Rf&=-9NeMJbm6w3Uo&{m&1SQ82;1N4)lYQV3h75_4KuC z_vnA?fc^Ja-LoUUAsl4+5hwVI*TA2j|NkqML-wo4ecslvZd+2j__sie3P#!7~6Sxqqndcb9zi(q2niT?MT!6zDp}F$eRa+&V`RI-5r^F3! zJyaCfe`|$Gp7;s{t_KK{7yQ=0iH<}`ghCERSTMhe_*CNeyfmIzdbNj_1md?u!k@{A zzW~4jp0YxlH2%iyZSS8biN9IRN(YD$WDoRE^N108nBlh}3aRlre+r>jXApj$_mvT1 z&|2(LBaD&ase-0z>pKn%w+t=9F9?Jhevg)$L4nlNX$>k`YLMJoC=@S%OMlvkXeY}g)7b@Ukjk_;D`&Cua;dJnxtEuFssG-OBv}mWkUgzhoBsea|Q5;vw0E16Fd<(cS$G4yfI2AO6 z*6aZuh5fl&Xg{5|X2rxLbSxgI8 zW}dHn@tFbFinN|qw$)mL)I}G41%_aHH?*O?}9o?MOYwy-2=(?QjsMCX=0T) zu3jZ)OKGing?Qa7Hr;&j%^ggK2I=us(Uaeee&rD6Cc_4fW=0CLoP&s4UOC}RsM7-I zo+t=%`G_j@3KT(@_OGdcd&f$=Wr~R?gSV^FJ)I52V-j{3QzjPqbe`hhne5&V#j^!FqK-`{^*rjjYQKD%VTIE&# zusK_)7~%0UY33A1I1^e2vpNbYE7lMK0DR6BgKo11STIbH$3RoF_eq?bIjlBE#5032 zA6MMn_7z@JelilxXmdkhC&~YR(xQn(9QP5^4ykS+Az970CYF1q-c$OX>0;fRu8Hd6urEmnZ^+p2?_NUBLhejcncMu^RR4;wh zW`}NznH}}+H8EsNvJ00Omapq!jSl7pmgEvQ1bPiTd;JSOLK{cm7m|3o$hHv(G?U|! zBqJelS7-ig=A^mY8~}0=2q&RsiM|&aaQo(30xZ@9h+G z>ALE_*&iU9&6mUd#8_QcvjmP6vm#%FLFxzv5qvwDY)Eoat6d|f3t(rt{e`dnsEuZU z6GRSShpS4G9+2$b|KUJ8#dydWJ7Dz8>a6OIAv=!GsRYzim$1h(12{K~fbCTi%IcVM zbaDend6LE816eyO29xQ5(U*fJ!6OTKH(kxB>yaeE5N*a9E?^vT_N?n?%&s6pxq1_; zq_ZwQep|FeZO~Gefk!Ka`kfMvI#@A|EZNL&Jx03OcZ3OHcGM(jA7|EkL7NuqiU62u z5GfkpR3AivFVz-(&J&YbLKKA`=c|)OUhG1E=TdFS)Cw*JPSM3`D6M0tSbHbp5l-~7 zUrT+kq_TkjS`#@IeZ^UkGk-K)Kzwa+Rh_O1md+zv+GO|GJ*$2@%)J?;@hZzg0#NDA!J|LRSB~=6tI_yFieZ%L#-8rQd6085{ou+=U(aR*+2gkx!J*3 zY0W!}<2y-6h}uTSn}0kdZK^d8!Spa{8V(aN3myyAMy$8>7iwY}^Mmu3zC4yT5&aQp z80y~x>ya60(d)Q*6G$)-nDFoQl~BGQ;%vOL33d42YG-?QsST%L z;E3+qexX*q;PjUCCxy&}^y$>PBPhAeyG^gP%rZb^KAkT}Ytu?vsTk->+x+CI#b(7< z3sjGoV&0a+!gY3DNk4JdYwqp-Ynsm^Wf;d? znM1qiVZP%~leIju8E?hzw`Ly2(ReL4C1Ciyh44+`v)Ep~lAZDl;)Wex-)BfBE2J3!JL%t@u(&W{h%e0)bDgf{R)`REgy6xBcwbsiWl=-se}73)G%P zU}`e(6GyK7C_+GQ+Ws3KiC&Oko&!yFU+}f;BP|gZbxYP;v9Xq35e(t6i_jZ`4A?LZ zQOb44aACO5O?{7bvZ{du%@%Xg)Zo0`1HM-!K<^;7*o)DnIee4Pt|4A2evrV7rlXOK zay0}RS`BnQOF23{pGuD}plPQbSA>16TE}SWrjsm+8tMha#lYEp?ed5sZEbTgWA}O# zwMKRmg3!^i4M)Xqa9B5AsDuJI1&)qkZ|Y^56CV*XXx%COu69A0QoOw;PO@>W+2%Oy zEHCap0xcZ9kx!%Z)5;(%Osv^Ey~9K!+(d#j`6Wq;zT?q&FHI5tlpssyI^P_(?f^KLLoSSARKNY3p__d%Id?l zR!r5mxsy%A{j_({%n7Jei55w{wFzziKJM$yaB_PxsCa8xjVjVg+@fWU^*6e~^`uB$ z`{q45b31ViX?($T1pa_N`qAZhZ;WIUj(~GhuPwC1abit{#+{)X=?Zn4bt|t(EG{-& z6dJaptX_tqlBb_cf;r#Djc;@uSxqNtB*%td@CFo`8WBr(?t4Gz3IA>~=nTV&bLCx8 zcwJ{p**>6!pm`sC&;1t7R2@Vi+_StT{6@PkFG@6aycu9+=xegD9Y-Yir!%%k6&Bue z#o&hIi8sE0XCfS#>46(VjiS?RrV*2VmM(p=%$`7bS1LUXn1?2G3tZo?)-^iezbzL1 z#8z%k%{Od%f(J3_vyNmg?5G zFK>T<@inEnDRwn`AXFl>7;VlKiYMn3xK2A#JnX;A$)|iKnNh-p3Rf*iQ&(VrZVK+f zf*T8|c+N zLS1>IGnEQx6BR;#<(R*U!UkIAlA}rErgcZ(zjC12JC;TBaCstO=`*C1BJ5I-q(^az zt!ef#ZgQ)0GT|xF`c1f_Tkv=`(%VKaD1%Ds*?a_Jtz}tdC&3v%fVet>cV4wT%ld-r zy{j9CuxN9Bzxix*&!l}@&)p8rwHMTRwx>UqT`z!e+J z-!35*t&M9fz6VP15apf0)V zg=PI@!}#x3i>-?OF{U|W*pf$x>A}`-VC>Yc!uZbZM`zcYtxDv0Lm!0VkpZ1K1Wm&@ z%;j=eXEpQyD+35uD(M{cJj|KnvWMRnNGg8GGYTB7=XM`6W)8%*Uc>Ca)cd$>orT7W z0lmy~=L$_m(fV@lVD{s&ooa+3itv5k&93hv*x>qWR823@9_s_iFlA;f)H&Nj2mJ}H zX6ZG0IcI6{G}E^WBfHKMcf!#7TcsTp1##dZrtXIGzk=fz!XPsyj?Fah!jE`?y8>Yo z2xKrIE--NEz6T=aa1Xf8rM1@vj;;+J#9ZTi1)R5=)nLMRzixmw6!dtqG+ll(NRjG5 zDZTKU+D|RGl){F7gYi*E=T@ zQ#%5?`y4gQ49;OXZ@OKRq5} z`zQ}F`RUTM!L2lPPw%MUi=S$yH#pSG{2@&y0wC9CR+xTicX6K4R57A4M>v z%cXGU!l>cg`F&m6!yjCNd^t8XY5`r#F@;!~mwwoggh9+gyZZWK3!^kwJeF7?Sf3z> zL{#~XdjGY{t(3cQ?103^Q=4& z@#TN0(vo3+q^Qs44V>YeP}KX?bL~n_Zz9cguu{mf+gj9mu@15mHZS=Lha#NlM>eSE zC9cAU8_zJR7&Ax$Ec^3{cdCw{4jX|@iFAv-Sn5$O1kgP~>ShjE-brAZdAiL1ZqLJ6e9O0fM&U~tp9}q`k z2%;L=J>a5OBTfft9?Ep_95xn9&dvW$y}!ksE6Dl_oMA7YTv?3kTd-PtBM%pO10rmU zNDp^vJicW*?f}I)R7liW27m)uwz8EK?Sn?h$)(D_bFC7l1ABI<*m)!;V#aF*qdeXl zlH&4?-ryNo{J2Exk>hChtNCyDh4j;!B0LgGxK6yaX43{9i@?zmhJ==Cre3zP5i*UJ z98aN>*Nw-vpN8DS7bsi{gb8t)Zqfx>2e)s`uhhP{-9-%TT33QI!lljsk@Iw4$1Kmu zM1m*D3ypKwl&(cD7}-!L;uen8Wk!>?=M{Ls(GV<ro#1}V8V z+>bew4soD?#)b%Me7D?{p8P4!;cJ;(a00*7u`K~7b=fe(430>+!6>s`T3CnaGmU6o zQ?xTrJb8iT z-l(A?c0feaF6z|N0nzH>k&E>BFR<%}AodkymxXp|g}lIk68GCJ!I#Z6T*nvgPPcd- zM=rB@ugE7a5$I+5A?q$rRm?>{h$E1zr1d%g-yIw# zpLg_Xg|}?{`ADn(rtiPN_ePlwk#ABj4J|_K`2wIBP2I8 zo?77Y5d1Xx7wUW2e37QC`V}1Kab+n}(0lFx;G{h_m=*OB)%_6I5Zj~veM#)uH+e{< z`SlOVd=1EdkBP)_QVyRYp~dwm)2SYr%fyGTFT^X+h#j!23YWhXECqhoRl0HN6JqTA z@m2)9`m~0mmTsa5Lbdb62_fwGXB5lF09;f`l9&sXg?z`r1<681-*9P$t4=UNe?5({ z0E5^pj9E(bM}Ib>D}y*^!u;(=@9&}%wB@EHBM7(39btERM12VLOj=a745OZa;A%|2 zHEFzHG1xd3bef(RRppda408oU)x|9Mp7bx;4U`;I&($P3L!#SPPtvFwYk82Q645Er z6=5D~@pq`#oW>eus^tMBZwMgu>uY=bCJjjTf!k~&IZGZq53hHH?kiVbeL4iunA(@X z!kppEe(vs!Sb0n*QXUtg4)7}`H8>NZqBxwvs<}B^1gdN|fjpRN`Jzg|E<-y2o?zte zuHZ^Qv8#eu(D{1h0;K|AfJZj-j2}FjV~3e{-OCq!q=K0E=Xd+xWH7Q;%20i^ zvARFc#dAmfS#5R1v-i631!n|t;e7r#o(X=Mn$qeLd^WvhDDbUb31}l7x$5R$9cbGNz#+RHcoH>0)oU5?EG7RsUvADP>^(obr`+KTXT-9zb zk^0GujT<6oFWwCpjbs#QynixGqg4H&BbgNM*SlwQejHa&Fx~r_Or7Pa-o4HM%U3Vo ztt?M=gQlHK;Ngg!a-7#uY(UzaPpLpOZT#u48+A8XmwRtHS^C9%+wwX-w%UJ7$XKIv z%CRSZ=uyyfoRYY7oQ)lDeVEBI#+jL8KD(}JDSAv9%6ct~YOhX__B-rOuuw;D>AmgT z_9ANNG-bc07O;nim;^=5q4eMUQ58(8N`@bcpSa!QPHoZ`A{tn9MJSERfA&kcadA!X zQZ$>w`8ORxWj*6}g1@oJd(P?ASIxAM(0mLRds}@-j*lsULw!d_@i65_yL?u%@!hpsle?V()}_?E!6na@V38Xl+Q+@6u-(+^08d^0ZH12%vL)SYbgQt z(PnQvre%uGC?*G7maaFEsZyGP_HkR)4Z|eZkp5T zCYL(K-!IBnwss6j9S&F6tH#I>e`w%{8o65Z`0{f$A&&Lu=RWw-{WIPFdm4%v1JN*N>N*k0?h!D9bizk!aUhxY6F(PN*g zb%^7I^xF^phn?{62h4>L6UWyOnN|9k=9!e}7I~I?&JSnqvHbGJd-<~0?vAgl(^sVu zw~3Cq1h3?u-`P?j;%pcyu8ynmJ?~?nRPZ|D_QC(W;gGNRom5hCkvU)ZI4vzL*SPca zsZ&XZZb;ze?(cE@T>OG9VfEiXBA$3#Io@Ywil6uNUBia{^Yq;_jIr{r zV_FT9sy^_#)19B9{`yym76t^2K5sbgzhg2sHr6;&AJdd9V#eTmad&6?+tS2GtL=^D z$P24`f4;XNf8dwcY^TxTBTJ#!5H804S1SntmiF{g?0>mNQCp!^i{q_Lu-_WHa*dlF zp1R{VLfNz$7sIZ23!c_RG1K?@=`*93*6YGM+ZxV|g?(_GT6ueyHa9n;RE6zHO`?#| zQKiS@7doFB98+5%(!?(e)r7%cFYG3`i5hO@*a`oqlkDH~_lHAHiJm(474=^a%pX3! zCvc2}1bau|2^sn?SKyycQR4Wn|MS8B@xQd)4G6Gw<(A_7>w)~^QLUuZ!Sd^`Lvovz zI`&FlHV^JkZ~2#FhQgoU`*Gp9q;CJ~uS0EnT1ly(B>pz_ z&&Bvlznz1_?!cek@;}bN&-rd4L*b7a=+a{TIQxJ7G!ks?Uw`0F!*NfRh^S#q=j5Nw@C)<(&t~|A<@e8K_{)v< ruVwbn&F~9{-#<6QFWiQbymRa%b??Wnq^TX-1OKS2oW|uTUkv(RS0_@s literal 0 HcmV?d00001 diff --git a/notebooks/community/matching_engine/matching_engine_for_indexing.ipynb b/notebooks/community/matching_engine/matching_engine_for_indexing.ipynb new file mode 100644 index 000000000..144eb8d6c --- /dev/null +++ b/notebooks/community/matching_engine/matching_engine_for_indexing.ipynb @@ -0,0 +1,1770 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ur8xi4C7S06n" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "JAPoU8Sm5E6e" + }, + "source": [ + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " Run in Google Cloud Notebooks\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tvgnzT1CKxrO" + }, + "source": [ + "## Overview\n", + "\n", + "This example demonstrates how to use the GCP ANN Service. It is a high scale, low latency solution, to find similar vectors (or more specifically \"embeddings\") for a large corpus. Moreover, it is a fully managed offering, further reducing operational overhead. It is built upon [Approximate Nearest Neighbor (ANN) technology](https://ai.googleblog.com/2020/07/announcing-scann-efficient-vector.html) developed by Google Research.\n", + "\n", + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [GloVe dataset](https://nlp.stanford.edu/projects/glove/).\n", + "\n", + "### Objective\n", + "\n", + "In this notebook, you will learn how to create Approximate Nearest Neighbor (ANN) Index, query against indexes, and validate the performance of the index. \n", + "\n", + "The steps performed include:\n", + "\n", + "* Create ANN Index and Brute Force Index\n", + "* Create an IndexEndpoint with VPC Network\n", + "* Deploy ANN Index and Brute Force Index\n", + "* Perform online query\n", + "* Compute recall\n", + "\n", + "\n", + "### Costs \n", + "\n", + "This tutorial uses billable components of Google Cloud:\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "S5zc4kbEiYCm" + }, + "source": [ + "## Before you begin\n", + "\n", + "* **Prepare a VPC network**. To reduce any network overhead that might lead to unnecessary increase in overhead latency, it is best to call the ANN endpoints from your VPC via a direct [VPC Peering](https://cloud.google.com/vertex-ai/docs/general/vpc-peering) connection. The following section describes how to setup a VPC Peering connection if you don't have one. This is a one-time initial setup task. You can also reuse existing VPC network and skip this section.\n", + "* **WARNING:** The match service gRPC API (to create online queries against your deployed index) has to be executed in a Google Cloud Notebook instance that is created with the following requirements:\n", + " * **In the same region as where your ANN service is deployed** (for example, if you set `REGION = \"us-central1\"` as same as the tutorial, the notebook instance has to be in `us-central1`).\n", + " * **Make sure you select the VPC network you created for ANN service** (instead of using the \"default\" one). That is, you will have to create the VPC network below and then create a new notebook instance that uses that VPC. \n", + " * If you run it in the colab or a Google Cloud Notebook instance in a different VPC network or region, the gRPC API will fail to peer the network (InactiveRPCError)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "lW2LneA5mmmP" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"\" # @param {type:\"string\"}\n", + "NETWORK_NAME = \"ucaip-haystack-vpc-network\" # @param {type:\"string\"}\n", + "PEERING_RANGE_NAME = \"ucaip-haystack-range\"\n", + "\n", + "# Create a VPC network\n", + "! gcloud compute networks create {NETWORK_NAME} --bgp-routing-mode=regional --subnet-mode=auto --project={PROJECT_ID}\n", + "\n", + "# Add necessary firewall rules\n", + "! gcloud compute firewall-rules create {NETWORK_NAME}-allow-icmp --network {NETWORK_NAME} --priority 65534 --project {PROJECT_ID} --allow icmp\n", + "\n", + "! gcloud compute firewall-rules create {NETWORK_NAME}-allow-internal --network {NETWORK_NAME} --priority 65534 --project {PROJECT_ID} --allow all --source-ranges 10.128.0.0/9\n", + "\n", + "! gcloud compute firewall-rules create {NETWORK_NAME}-allow-rdp --network {NETWORK_NAME} --priority 65534 --project {PROJECT_ID} --allow tcp:3389\n", + "\n", + "! gcloud compute firewall-rules create {NETWORK_NAME}-allow-ssh --network {NETWORK_NAME} --priority 65534 --project {PROJECT_ID} --allow tcp:22\n", + "\n", + "# Reserve IP range\n", + "! gcloud compute addresses create {PEERING_RANGE_NAME} --global --prefix-length=16 --network={NETWORK_NAME} --purpose=VPC_PEERING --project={PROJECT_ID} --description=\"peering range for uCAIP Haystack.\"\n", + "\n", + "# Set up peering with service networking\n", + "! gcloud services vpc-peerings connect --service=servicenetworking.googleapis.com --network={NETWORK_NAME} --ranges={PEERING_RANGE_NAME} --project={PROJECT_ID}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "d3uj8x73nDX_" + }, + "source": [ + "* Authentication: `$ gcloud auth login` rerun this in Google Cloud Notebook terminal when you are logged out and need the credential again." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "i7EUnXsZhAGF" + }, + "source": [ + "### Installation\n", + "\n", + "Download and install the latest (preview) version of the Vertex SDK for Python." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wyy5Lbnzg5fi" + }, + "outputs": [], + "source": [ + "! pip install -U git+https://github.com/googleapis/python-aiplatform.git@main-test --user" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "irSMQn6gZ19l" + }, + "source": [ + "Install the `h5py` to prepare sample dataset, and the `grpcio-tools` for querying against the index. " + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "-h5sqwOEZ5Yq" + }, + "outputs": [], + "source": [ + "! pip install -U grpcio-tools --user\n", + "! pip install -U h5py --user" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "hhq5zEbGg0XX" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "After you install the additional packages, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "EzrelQZ22IZj" + }, + "outputs": [], + "source": [ + "# Automatically restart kernel after installs\n", + "import os\n", + "\n", + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "BF1j6f9HApxa" + }, + "source": [ + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager).\n", + "\n", + "1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n", + "\n", + "1. [Enable the Vertex AI API and Compute Engine API, and Service Networking API](https://console.cloud.google.com/flows/enableapi?apiid=aiplatform.googleapis.com,compute_component,servicenetworking.googleapis.com).\n", + "\n", + "1. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "WReHDGG5g0XY" + }, + "source": [ + "#### Set your project ID\n", + "\n", + "**If you don't know your project ID**, you may be able to get your project ID using `gcloud`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "oM1iC_MfAts1" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "PROJECT_ID = \"\"\n", + "\n", + "# Get your Google Cloud project ID from gcloud\n", + "if not os.getenv(\"IS_TESTING\"):\n", + " shell_output=!gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID: \", PROJECT_ID)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "qJYoRfYng0XZ" + }, + "source": [ + "Otherwise, set your project ID here." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "riG_qUokg0XZ" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None:\n", + " PROJECT_ID = \"\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "zgPO1eR3CYjk" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all\n", + "Cloud Storage buckets.\n", + "\n", + "You may also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Make sure to [choose a region where Vertex AI services are\n", + "available](https://cloud.google.com/vertex-ai/docs/general/locations#available_regions). You may\n", + "not use a Multi-Regional Storage bucket for training with Vertex AI." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "MzGDU7TWdts_" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}\n", + "REGION = \"us-central1\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cf221059d072" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")\n", + "\n", + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "-EcIXiGsCePi" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "NIq7R4HZCfIc" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ucvCsknMCims" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "vhOb7YnwClBb" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "XoEqT2Y4DJmf" + }, + "source": [ + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Y9Uo3tifg1kx" + }, + "source": [ + "Import the Vertex AI (unified) client library into your Python environment. \n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "f2d05ab4126a" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "import grpc\n", + "import h5py\n", + "from google.cloud import aiplatform_v1beta1\n", + "from google.protobuf import struct_pb2" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pRUOFELefqf1" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\"\n", + "ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "NETWORK_NAME = \"ucaip-haystack-vpc-network\" # @param {type:\"string\"}\n", + "\n", + "\n", + "AUTH_TOKEN = !gcloud auth print-access-token\n", + "PROJECT_NUMBER = !gcloud projects list --filter=\"PROJECT_ID:'{PROJECT_ID}'\" --format='value(PROJECT_NUMBER)'\n", + "PROJECT_NUMBER = PROJECT_NUMBER[0]\n", + "\n", + "PARENT = \"projects/{}/locations/{}\".format(PROJECT_ID, REGION)\n", + "\n", + "print(\"ENDPOINT: {}\".format(ENDPOINT))\n", + "print(\"PROJECT_ID: {}\".format(PROJECT_ID))\n", + "print(\"REGION: {}\".format(REGION))\n", + "\n", + "!gcloud config set project {PROJECT_ID}\n", + "!gcloud config set ai_platform/region {REGION}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "lR6Wwv-hCCN-" + }, + "source": [ + "## Prepare the Data\n", + "\n", + "The GloVe dataset consists of a set of pre-trained embeddings. The embeddings are split into a \"train\" split, and a \"test\" split.\n", + "We will create a vector search index from the \"train\" split, and use the embedding vectors in the \"test\" split as query vectors to test the vector search index.\n", + "\n", + "NOTE: While the data split uses the term \"train\", these are pre-trained embeddings and thus are ready to be indexed for search. The terms \"train\" and \"test\" split are used just to be consistent with usual machine learning terminology.\n", + "\n", + "Download the GloVe dataset.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "9wzS85TeB9dG" + }, + "outputs": [], + "source": [ + "! gsutil cp gs://cloud-samples-data/ai-platform-unified/matching_engine/glove-100-angular.hdf5 ." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "4fAO9CMoCNtq" + }, + "source": [ + "Read the data into memory.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "lZ3JQTS6CN-3" + }, + "outputs": [], + "source": [ + "# The number of nearest neighbors to be retrieved from database for each query.\n", + "k = 10\n", + "\n", + "h5 = h5py.File(\"glove-100-angular.hdf5\", \"r\")\n", + "train = h5[\"train\"]\n", + "test = h5[\"test\"]" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pE6bBBo7GjJK" + }, + "outputs": [], + "source": [ + "train[0]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aQIQSyF9GtSv" + }, + "source": [ + "Save the train split in JSONL format.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "18wCiTwfG40P" + }, + "outputs": [], + "source": [ + "with open(\"glove100.json\", \"w\") as f:\n", + " for i in range(len(train)):\n", + " f.write('{\"id\":\"' + str(i) + '\",')\n", + " f.write('\"embedding\":[' + \",\".join(str(x) for x in train[i]) + \"]}\")\n", + " f.write(\"\\n\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "QuVl8DrWG8NS" + }, + "source": [ + "Upload the training data to GCS." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Gk6YmPMoG8aX" + }, + "outputs": [], + "source": [ + "# NOTE: Everything in this GCS DIR will be DELETED before uploading the data.\n", + "\n", + "! gsutil rm -rf {BUCKET_NAME}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "3PgsA_vbI8Vg" + }, + "outputs": [], + "source": [ + "! gsutil cp glove100.json {BUCKET_NAME}/glove100.json" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "3RX6g7FaJFes" + }, + "outputs": [], + "source": [ + "! gsutil ls {BUCKET_NAME}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "mglUPwHpJH98" + }, + "source": [ + "## Create Indexes\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "qhIBCQ7dDSbW" + }, + "source": [ + "### Create ANN Index (for Production Usage)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "DDAvm_mj_BVs" + }, + "outputs": [], + "source": [ + "index_client = aiplatform_v1beta1.IndexServiceClient(\n", + " client_options=dict(api_endpoint=ENDPOINT)\n", + ")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "qiIg9b5zJLi1" + }, + "outputs": [], + "source": [ + "DIMENSIONS = 100\n", + "DISPLAY_NAME = \"glove_100_1\"\n", + "DISPLAY_NAME_BRUTE_FORCE = DISPLAY_NAME + \"_brute_force\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "svLYiDf0OD2G" + }, + "source": [ + "Create the ANN index configuration:\n", + "\n", + "Please read the documentation to understand the various configuration parameters that can be used to tune the index\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Tfa8IoNrOCJh" + }, + "outputs": [], + "source": [ + "treeAhConfig = struct_pb2.Struct(\n", + " fields={\n", + " \"leafNodeEmbeddingCount\": struct_pb2.Value(number_value=500),\n", + " \"leafNodesToSearchPercent\": struct_pb2.Value(number_value=7),\n", + " }\n", + ")\n", + "\n", + "algorithmConfig = struct_pb2.Struct(\n", + " fields={\"treeAhConfig\": struct_pb2.Value(struct_value=treeAhConfig)}\n", + ")\n", + "\n", + "config = struct_pb2.Struct(\n", + " fields={\n", + " \"dimensions\": struct_pb2.Value(number_value=DIMENSIONS),\n", + " \"approximateNeighborsCount\": struct_pb2.Value(number_value=150),\n", + " \"distanceMeasureType\": struct_pb2.Value(string_value=\"DOT_PRODUCT_DISTANCE\"),\n", + " \"algorithmConfig\": struct_pb2.Value(struct_value=algorithmConfig),\n", + " }\n", + ")\n", + "\n", + "metadata = struct_pb2.Struct(\n", + " fields={\n", + " \"config\": struct_pb2.Value(struct_value=config),\n", + " \"contentsDeltaUri\": struct_pb2.Value(string_value=BUCKET_NAME),\n", + " }\n", + ")\n", + "\n", + "ann_index = {\n", + " \"display_name\": DISPLAY_NAME,\n", + " \"description\": \"Glove 100 ANN index\",\n", + " \"metadata\": struct_pb2.Value(struct_value=metadata),\n", + "}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "xzY7TpUSJcTV" + }, + "outputs": [], + "source": [ + "ann_index = index_client.create_index(parent=PARENT, index=ann_index)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "oLBD2xXG_tv7" + }, + "outputs": [], + "source": [ + "# Poll the operation until it's done successfullly.\n", + "# This will take ~45 min.\n", + "\n", + "while True:\n", + " if ann_index.done():\n", + " break\n", + " print(\"Poll the operation to create index...\")\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "17jrQi501QyX" + }, + "outputs": [], + "source": [ + "INDEX_RESOURCE_NAME = ann_index.result().name\n", + "INDEX_RESOURCE_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "kSsqZuyoA1SG" + }, + "source": [ + "### Create Brute Force Index (for Ground Truth)\n", + "\n", + "The brute force index uses a naive brute force method to find the nearest neighbors. This method is not fast or efficient. Hence brute force indices are not recommended for production usage. They are to be used to find the \"ground truth\" set of neighbors, so that the \"ground truth\" set can be used to measure recall of the indices being tuned for production usage. To ensure an apples to apples comparison, the `distanceMeasureType` and `featureNormType`, `dimensions` of the brute force index should match those of the production indices being tuned.\n", + "\n", + "Create the brute force index configuration:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "5ExrilcZA87V" + }, + "outputs": [], + "source": [ + "from google.protobuf import *\n", + "\n", + "algorithmConfig = struct_pb2.Struct(\n", + " fields={\"bruteForceConfig\": struct_pb2.Value(struct_value=struct_pb2.Struct())}\n", + ")\n", + "\n", + "config = struct_pb2.Struct(\n", + " fields={\n", + " \"dimensions\": struct_pb2.Value(number_value=DIMENSIONS),\n", + " \"approximateNeighborsCount\": struct_pb2.Value(number_value=150),\n", + " \"distanceMeasureType\": struct_pb2.Value(string_value=\"DOT_PRODUCT_DISTANCE\"),\n", + " \"algorithmConfig\": struct_pb2.Value(struct_value=algorithmConfig),\n", + " }\n", + ")\n", + "\n", + "metadata = struct_pb2.Struct(\n", + " fields={\n", + " \"config\": struct_pb2.Value(struct_value=config),\n", + " \"contentsDeltaUri\": struct_pb2.Value(string_value=BUCKET_NAME),\n", + " }\n", + ")\n", + "\n", + "brute_force_index = {\n", + " \"display_name\": DISPLAY_NAME_BRUTE_FORCE,\n", + " \"description\": \"Glove 100 index (brute force)\",\n", + " \"metadata\": struct_pb2.Value(struct_value=metadata),\n", + "}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "DXnBLqjXBsv8" + }, + "outputs": [], + "source": [ + "brute_force_index = index_client.create_index(parent=PARENT, index=brute_force_index)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "AtwELX4Peq2n" + }, + "outputs": [], + "source": [ + "# Poll the operation until it's done successfullly.\n", + "# This will take ~45 min.\n", + "\n", + "while True:\n", + " if brute_force_index.done():\n", + " break\n", + " print(\"Poll the operation to create index...\")\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "_oD5SieYJbbW" + }, + "outputs": [], + "source": [ + "INDEX_BRUTE_FORCE_RESOURCE_NAME = brute_force_index.result().name\n", + "INDEX_BRUTE_FORCE_RESOURCE_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "qV2xjAnDDObD" + }, + "source": [ + "## Create an IndexEndpoint with VPC Network" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "2AWeQ6e04m36" + }, + "outputs": [], + "source": [ + "index_endpoint_client = aiplatform_v1beta1.IndexEndpointServiceClient(\n", + " client_options=dict(api_endpoint=ENDPOINT)\n", + ")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "BpZQoJyxDlbO" + }, + "outputs": [], + "source": [ + "VPC_NETWORK_NAME = \"projects/{}/global/networks/{}\".format(PROJECT_NUMBER, NETWORK_NAME)\n", + "VPC_NETWORK_NAME" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "mColTdkIDoVZ" + }, + "outputs": [], + "source": [ + "index_endpoint = {\n", + " \"display_name\": \"index_endpoint_for_demo\",\n", + " \"network\": VPC_NETWORK_NAME,\n", + "}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "QuARXzJVGyQX" + }, + "outputs": [], + "source": [ + "r = index_endpoint_client.create_index_endpoint(\n", + " parent=PARENT, index_endpoint=index_endpoint\n", + ")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "YRL1AYF_HpZR" + }, + "outputs": [], + "source": [ + "r.result()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "PJ3bcZqi-cfM" + }, + "outputs": [], + "source": [ + "INDEX_ENDPOINT_NAME = r.result().name\n", + "INDEX_ENDPOINT_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "np2cgVuuIe9k" + }, + "source": [ + "## Deploy Indexes" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "8Ew1UgcIIiJG" + }, + "source": [ + "### Deploy ANN Index" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "nLOYTGygIlMK" + }, + "outputs": [], + "source": [ + "DEPLOYED_INDEX_ID = \"ann_glove_deployed\"" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "M-W5LYQrKTzi" + }, + "outputs": [], + "source": [ + "deploy_ann_index = {\n", + " \"id\": DEPLOYED_INDEX_ID,\n", + " \"display_name\": DEPLOYED_INDEX_ID,\n", + " \"index\": INDEX_RESOURCE_NAME,\n", + "}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "_uK4WOgqN1NG" + }, + "outputs": [], + "source": [ + "r = index_endpoint_client.deploy_index(\n", + " index_endpoint=INDEX_ENDPOINT_NAME, deployed_index=deploy_ann_index\n", + ")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "2Lt0jvsSeekz" + }, + "outputs": [], + "source": [ + "# Poll the operation until it's done successfullly.\n", + "\n", + "while True:\n", + " if r.done():\n", + " break\n", + " print(\"Poll the operation to deploy index...\")\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "8ajRpqe2J9aS" + }, + "outputs": [], + "source": [ + "r.result()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "RNZnXmO5AhDO" + }, + "source": [ + "### Deploy Brute Force Index" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "3p9e4828AkSv" + }, + "outputs": [], + "source": [ + "DEPLOYED_BRUTE_FORCE_INDEX_ID = \"glove_brute_force_deployed\"" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "6PgQKgHQAq3p" + }, + "outputs": [], + "source": [ + "deploy_brute_force_index = {\n", + " \"id\": DEPLOYED_BRUTE_FORCE_INDEX_ID,\n", + " \"display_name\": DEPLOYED_BRUTE_FORCE_INDEX_ID,\n", + " \"index\": INDEX_BRUTE_FORCE_RESOURCE_NAME,\n", + "}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "-2kgd01SA4rk" + }, + "outputs": [], + "source": [ + "r = index_endpoint_client.deploy_index(\n", + " index_endpoint=INDEX_ENDPOINT_NAME, deployed_index=deploy_brute_force_index\n", + ")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "R6nZiQP-c2nu" + }, + "outputs": [], + "source": [ + "# Poll the operation until it's done successfullly.\n", + "\n", + "while True:\n", + " if r.done():\n", + " break\n", + " print(\"Poll the operation to deploy index...\")\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "2v7J36ShA9Sw" + }, + "outputs": [], + "source": [ + "r.result()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "6LCGvBNvBd8D" + }, + "source": [ + "## Create Online Queries\n", + "\n", + "After you built your indexes, you may query against the deployed index through the online querying gRPC API (Match service) within the virtual machine instances from the same region (for example 'us-central1' in this tutorial). \n", + "\n", + "The way a client uses this gRPC API is by folowing steps:\n", + "\n", + "* Write `match_service.proto` locally\n", + "* Clone the repository that contains the dependencies of match_service.proto in the Terminal:\n", + "\n", + "`$ mkdir third_party && cd third_party`\n", + "\n", + "`$ git clone https://github.com/googleapis/googleapis.git`\n", + "\n", + "* Compile the protocal buffer (see below)\n", + "* Obtain the index endpoint\n", + "* Use a code-generated stub to make the call, passing the parameter values" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "cellView": "code", + "id": "rUUW3ViQE88D" + }, + "outputs": [], + "source": [ + "%%writefile match_service.proto\n", + "\n", + "syntax = \"proto3\";\n", + "\n", + "package google.cloud.aiplatform.container.v1beta1;\n", + "\n", + "import \"google/rpc/status.proto\";\n", + "\n", + "// MatchService is a Google managed service for efficient vector similarity\n", + "// search at scale.\n", + "service MatchService {\n", + " // Returns the nearest neighbors for the query. If it is a sharded\n", + " // deployment, calls the other shards and aggregates the responses.\n", + " rpc Match(MatchRequest) returns (MatchResponse) {}\n", + "\n", + " // Returns the nearest neighbors for batch queries. If it is a sharded\n", + " // deployment, calls the other shards and aggregates the responses.\n", + " rpc BatchMatch(BatchMatchRequest) returns (BatchMatchResponse) {}\n", + "}\n", + "\n", + "// Parameters for a match query.\n", + "message MatchRequest {\n", + " // The ID of the DeploydIndex that will serve the request.\n", + " // This MatchRequest is sent to a specific IndexEndpoint of the Control API,\n", + " // as per the IndexEndpoint.network. That IndexEndpoint also has\n", + " // IndexEndpoint.deployed_indexes, and each such index has an\n", + " // DeployedIndex.id field.\n", + " // The value of the field below must equal one of the DeployedIndex.id\n", + " // fields of the IndexEndpoint that is being called for this request.\n", + " string deployed_index_id = 1;\n", + "\n", + " // The embedding values.\n", + " repeated float float_val = 2;\n", + "\n", + " // The number of nearest neighbors to be retrieved from database for\n", + " // each query. If not set, will use the default from\n", + " // the service configuration.\n", + " int32 num_neighbors = 3;\n", + "\n", + " // The list of restricts.\n", + " repeated Namespace restricts = 4;\n", + "\n", + " // Crowding is a constraint on a neighbor list produced by nearest neighbor\n", + " // search requiring that no more than some value k' of the k neighbors\n", + " // returned have the same value of crowding_attribute.\n", + " // It's used for improving result diversity.\n", + " // This field is the maximum number of matches with the same crowding tag.\n", + " int32 per_crowding_attribute_num_neighbors = 5;\n", + "\n", + " // The number of neighbors to find via approximate search before\n", + " // exact reordering is performed. If not set, the default value from scam\n", + " // config is used; if set, this value must be > 0.\n", + " int32 approx_num_neighbors = 6;\n", + "\n", + " // The fraction of the number of leaves to search, set at query time allows\n", + " // user to tune search performance. This value increase result in both search\n", + " // accuracy and latency increase. The value should be between 0.0 and 1.0. If\n", + " // not set or set to 0.0, query uses the default value specified in\n", + " // NearestNeighborSearchConfig.TreeAHConfig.leaf_nodes_to_search_percent.\n", + " int32 leaf_nodes_to_search_percent_override = 7;\n", + "}\n", + "\n", + "// Response of a match query.\n", + "message MatchResponse {\n", + " message Neighbor {\n", + " // The ids of the matches.\n", + " string id = 1;\n", + "\n", + " // The distances of the matches.\n", + " double distance = 2;\n", + " }\n", + " // All its neighbors.\n", + " repeated Neighbor neighbor = 1;\n", + "}\n", + "\n", + "// Parameters for a batch match query.\n", + "message BatchMatchRequest {\n", + " // Batched requests against one index.\n", + " message BatchMatchRequestPerIndex {\n", + " // The ID of the DeploydIndex that will serve the request.\n", + " string deployed_index_id = 1;\n", + "\n", + " // The requests against the index identified by the above deployed_index_id.\n", + " repeated MatchRequest requests = 2;\n", + "\n", + " // Selects the optimal batch size to use for low-level batching. Queries\n", + " // within each low level batch are executed sequentially while low level\n", + " // batches are executed in parallel.\n", + " // This field is optional, defaults to 0 if not set. A non-positive number\n", + " // disables low level batching, i.e. all queries are executed sequentially.\n", + " int32 low_level_batch_size = 3;\n", + " }\n", + "\n", + " // The batch requests grouped by indexes.\n", + " repeated BatchMatchRequestPerIndex requests = 1;\n", + "}\n", + "\n", + "// Response of a batch match query.\n", + "message BatchMatchResponse {\n", + " // Batched responses for one index.\n", + " message BatchMatchResponsePerIndex {\n", + " // The ID of the DeployedIndex that produced the responses.\n", + " string deployed_index_id = 1;\n", + "\n", + " // The match responses produced by the index identified by the above\n", + " // deployed_index_id. This field is set only when the query against that\n", + " // index succeed.\n", + " repeated MatchResponse responses = 2;\n", + "\n", + " // The status of response for the batch query identified by the above\n", + " // deployed_index_id.\n", + " google.rpc.Status status = 3;\n", + " }\n", + "\n", + " // The batched responses grouped by indexes.\n", + " repeated BatchMatchResponsePerIndex responses = 1;\n", + "}\n", + "\n", + "// Namespace specifies the rules for determining the datapoints that are\n", + "// eligible for each matching query, overall query is an AND across namespaces.\n", + "message Namespace {\n", + " // The string name of the namespace that this proto is specifying,\n", + " // such as \"color\", \"shape\", \"geo\", or \"tags\".\n", + " string name = 1;\n", + "\n", + " // The allowed tokens in the namespace.\n", + " repeated string allow_tokens = 2;\n", + "\n", + " // The denied tokens in the namespace.\n", + " // The denied tokens have exactly the same format as the token fields, but\n", + " // represents a negation. When a token is denied, then matches will be\n", + " // excluded whenever the other datapoint has that token.\n", + " //\n", + " // For example, if a query specifies {color: red, blue, !purple}, then that\n", + " // query will match datapoints that are red or blue, but if those points are\n", + " // also purple, then they will be excluded even if they are red/blue.\n", + " repeated string deny_tokens = 3;\n", + "}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dfh48KLTJkaF" + }, + "source": [ + "Compile the protocol buffer, and then `match_service_pb2.py` and `match_service_pb2_grpc.py` are generated." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "EehHF_AeGmQT" + }, + "outputs": [], + "source": [ + "! python -m grpc_tools.protoc -I=. --proto_path=third_party/googleapis --python_out=. --grpc_python_out=. match_service.proto" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "8wXTSgz1Bl0x" + }, + "source": [ + "Obtain the Private Endpoint: " + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "zwA042IZJv1h" + }, + "outputs": [], + "source": [ + "DEPLOYED_INDEX_SERVER_IP = (\n", + " list(index_endpoint_client.list_index_endpoints(parent=PARENT))[0]\n", + " .deployed_indexes[0]\n", + " .private_endpoints.match_grpc_address\n", + ")\n", + "DEPLOYED_INDEX_SERVER_IP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "IcXa9lSuB9AT" + }, + "source": [ + "Test your query:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "zhgTqsI1spsH" + }, + "outputs": [], + "source": [ + "import match_service_pb2\n", + "import match_service_pb2_grpc\n", + "\n", + "channel = grpc.insecure_channel(\"{}:10000\".format(DEPLOYED_INDEX_SERVER_IP))\n", + "stub = match_service_pb2_grpc.MatchServiceStub(channel)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "A3KYVw5HB-4v" + }, + "outputs": [], + "source": [ + "# Test query\n", + "query = [\n", + " -0.11333,\n", + " 0.48402,\n", + " 0.090771,\n", + " -0.22439,\n", + " 0.034206,\n", + " -0.55831,\n", + " 0.041849,\n", + " -0.53573,\n", + " 0.18809,\n", + " -0.58722,\n", + " 0.015313,\n", + " -0.014555,\n", + " 0.80842,\n", + " -0.038519,\n", + " 0.75348,\n", + " 0.70502,\n", + " -0.17863,\n", + " 0.3222,\n", + " 0.67575,\n", + " 0.67198,\n", + " 0.26044,\n", + " 0.4187,\n", + " -0.34122,\n", + " 0.2286,\n", + " -0.53529,\n", + " 1.2582,\n", + " -0.091543,\n", + " 0.19716,\n", + " -0.037454,\n", + " -0.3336,\n", + " 0.31399,\n", + " 0.36488,\n", + " 0.71263,\n", + " 0.1307,\n", + " -0.24654,\n", + " -0.52445,\n", + " -0.036091,\n", + " 0.55068,\n", + " 0.10017,\n", + " 0.48095,\n", + " 0.71104,\n", + " -0.053462,\n", + " 0.22325,\n", + " 0.30917,\n", + " -0.39926,\n", + " 0.036634,\n", + " -0.35431,\n", + " -0.42795,\n", + " 0.46444,\n", + " 0.25586,\n", + " 0.68257,\n", + " -0.20821,\n", + " 0.38433,\n", + " 0.055773,\n", + " -0.2539,\n", + " -0.20804,\n", + " 0.52522,\n", + " -0.11399,\n", + " -0.3253,\n", + " -0.44104,\n", + " 0.17528,\n", + " 0.62255,\n", + " 0.50237,\n", + " -0.7607,\n", + " -0.071786,\n", + " 0.0080131,\n", + " -0.13286,\n", + " 0.50097,\n", + " 0.18824,\n", + " -0.54722,\n", + " -0.42664,\n", + " 0.4292,\n", + " 0.14877,\n", + " -0.0072514,\n", + " -0.16484,\n", + " -0.059798,\n", + " 0.9895,\n", + " -0.61738,\n", + " 0.054169,\n", + " 0.48424,\n", + " -0.35084,\n", + " -0.27053,\n", + " 0.37829,\n", + " 0.11503,\n", + " -0.39613,\n", + " 0.24266,\n", + " 0.39147,\n", + " -0.075256,\n", + " 0.65093,\n", + " -0.20822,\n", + " -0.17456,\n", + " 0.53571,\n", + " -0.16537,\n", + " 0.13582,\n", + " -0.56016,\n", + " 0.016964,\n", + " 0.1277,\n", + " 0.94071,\n", + " -0.22608,\n", + " -0.021106,\n", + "]\n", + "\n", + "request = match_service_pb2.MatchRequest()\n", + "request.deployed_index_id = DEPLOYED_INDEX_ID\n", + "for val in query:\n", + " request.float_val.append(val)\n", + "\n", + "response = stub.Match(request)\n", + "response" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "_mNwdU9_B_Ez" + }, + "source": [ + "### Batch Query\n", + "\n", + "You can run multiple queries in a single RPC call using the BatchMatch API:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "L55vcqox5cQz" + }, + "outputs": [], + "source": [ + "def get_request(embedding, deployed_index_id):\n", + " request = match_service_pb2.MatchRequest(num_neighbors=k)\n", + " request.deployed_index_id = deployed_index_id\n", + " for val in embedding:\n", + " request.float_val.append(val)\n", + " return request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "A3KYVw5HB-4v" + }, + "outputs": [], + "source": [ + "# Test query\n", + "queries = [\n", + " [\n", + " -0.11333,\n", + " 0.48402,\n", + " 0.090771,\n", + " -0.22439,\n", + " 0.034206,\n", + " -0.55831,\n", + " 0.041849,\n", + " -0.53573,\n", + " 0.18809,\n", + " -0.58722,\n", + " 0.015313,\n", + " -0.014555,\n", + " 0.80842,\n", + " -0.038519,\n", + " 0.75348,\n", + " 0.70502,\n", + " -0.17863,\n", + " 0.3222,\n", + " 0.67575,\n", + " 0.67198,\n", + " 0.26044,\n", + " 0.4187,\n", + " -0.34122,\n", + " 0.2286,\n", + " -0.53529,\n", + " 1.2582,\n", + " -0.091543,\n", + " 0.19716,\n", + " -0.037454,\n", + " -0.3336,\n", + " 0.31399,\n", + " 0.36488,\n", + " 0.71263,\n", + " 0.1307,\n", + " -0.24654,\n", + " -0.52445,\n", + " -0.036091,\n", + " 0.55068,\n", + " 0.10017,\n", + " 0.48095,\n", + " 0.71104,\n", + " -0.053462,\n", + " 0.22325,\n", + " 0.30917,\n", + " -0.39926,\n", + " 0.036634,\n", + " -0.35431,\n", + " -0.42795,\n", + " 0.46444,\n", + " 0.25586,\n", + " 0.68257,\n", + " -0.20821,\n", + " 0.38433,\n", + " 0.055773,\n", + " -0.2539,\n", + " -0.20804,\n", + " 0.52522,\n", + " -0.11399,\n", + " -0.3253,\n", + " -0.44104,\n", + " 0.17528,\n", + " 0.62255,\n", + " 0.50237,\n", + " -0.7607,\n", + " -0.071786,\n", + " 0.0080131,\n", + " -0.13286,\n", + " 0.50097,\n", + " 0.18824,\n", + " -0.54722,\n", + " -0.42664,\n", + " 0.4292,\n", + " 0.14877,\n", + " -0.0072514,\n", + " -0.16484,\n", + " -0.059798,\n", + " 0.9895,\n", + " -0.61738,\n", + " 0.054169,\n", + " 0.48424,\n", + " -0.35084,\n", + " -0.27053,\n", + " 0.37829,\n", + " 0.11503,\n", + " -0.39613,\n", + " 0.24266,\n", + " 0.39147,\n", + " -0.075256,\n", + " 0.65093,\n", + " -0.20822,\n", + " -0.17456,\n", + " 0.53571,\n", + " -0.16537,\n", + " 0.13582,\n", + " -0.56016,\n", + " 0.016964,\n", + " 0.1277,\n", + " 0.94071,\n", + " -0.22608,\n", + " -0.021106,\n", + " ],\n", + " [\n", + " -0.99544,\n", + " -2.3651,\n", + " -0.24332,\n", + " -1.0321,\n", + " 0.42052,\n", + " -1.1817,\n", + " -0.16451,\n", + " -1.683,\n", + " 0.49673,\n", + " -0.27258,\n", + " -0.025397,\n", + " 0.34188,\n", + " 1.5523,\n", + " 1.3532,\n", + " 0.33297,\n", + " -0.0056677,\n", + " -0.76525,\n", + " 0.49587,\n", + " 1.2211,\n", + " 0.83394,\n", + " -0.20031,\n", + " -0.59657,\n", + " 0.38485,\n", + " -0.23487,\n", + " -1.0725,\n", + " 0.95856,\n", + " 0.16161,\n", + " -1.2496,\n", + " 1.6751,\n", + " 0.73899,\n", + " 0.051347,\n", + " -0.42702,\n", + " 0.16257,\n", + " -0.16772,\n", + " 0.40146,\n", + " 0.29837,\n", + " 0.96204,\n", + " -0.36232,\n", + " -0.47848,\n", + " 0.78278,\n", + " 0.14834,\n", + " 1.3407,\n", + " 0.47834,\n", + " -0.39083,\n", + " -1.037,\n", + " -0.24643,\n", + " -0.75841,\n", + " 0.7669,\n", + " -0.37363,\n", + " 0.52741,\n", + " 0.018563,\n", + " -0.51301,\n", + " 0.97674,\n", + " 0.55232,\n", + " 1.1584,\n", + " 0.73715,\n", + " 1.3055,\n", + " -0.44743,\n", + " -0.15961,\n", + " 0.85006,\n", + " -0.34092,\n", + " -0.67667,\n", + " 0.2317,\n", + " 1.5582,\n", + " 1.2308,\n", + " -0.62213,\n", + " -0.032801,\n", + " 0.1206,\n", + " -0.25899,\n", + " -0.02756,\n", + " -0.52814,\n", + " -0.93523,\n", + " 0.58434,\n", + " -0.24799,\n", + " 0.37692,\n", + " 0.86527,\n", + " 0.069626,\n", + " 1.3096,\n", + " 0.29975,\n", + " -1.3651,\n", + " -0.32048,\n", + " -0.13741,\n", + " 0.33329,\n", + " -1.9113,\n", + " -0.60222,\n", + " -0.23921,\n", + " 0.12664,\n", + " -0.47961,\n", + " -0.89531,\n", + " 0.62054,\n", + " 0.40869,\n", + " -0.08503,\n", + " 0.6413,\n", + " -0.84044,\n", + " -0.74325,\n", + " -0.19426,\n", + " 0.098722,\n", + " 0.32648,\n", + " -0.67621,\n", + " -0.62692,\n", + " ],\n", + "]\n", + "\n", + "batch_request = match_service_pb2.BatchMatchRequest()\n", + "batch_request_ann = match_service_pb2.BatchMatchRequest.BatchMatchRequestPerIndex()\n", + "batch_request_brute_force = (\n", + " match_service_pb2.BatchMatchRequest.BatchMatchRequestPerIndex()\n", + ")\n", + "batch_request_ann.deployed_index_id = DEPLOYED_INDEX_ID\n", + "batch_request_brute_force.deployed_index_id = DEPLOYED_BRUTE_FORCE_INDEX_ID\n", + "for query in queries:\n", + " batch_request_ann.requests.append(get_request(query, DEPLOYED_INDEX_ID))\n", + " batch_request_brute_force.requests.append(\n", + " get_request(query, DEPLOYED_BRUTE_FORCE_INDEX_ID)\n", + " )\n", + "batch_request.requests.append(batch_request_ann)\n", + "batch_request.requests.append(batch_request_brute_force)\n", + "\n", + "response = stub.BatchMatch(batch_request)\n", + "response" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "_mNwdU9_B_Ez" + }, + "source": [ + "### Compute Recall\n", + "\n", + "Use deployed brute force Index as the ground truth to calculate the recall of ANN Index:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "L55vcqox5cQz" + }, + "outputs": [], + "source": [ + "def get_neighbors(embedding, deployed_index_id):\n", + " request = match_service_pb2.MatchRequest(num_neighbors=k)\n", + " request.deployed_index_id = deployed_index_id\n", + " for val in embedding:\n", + " request.float_val.append(val)\n", + " response = stub.Match(request)\n", + " return [int(n.id) for n in response.neighbor]" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "2vqQxfD9ufJm" + }, + "outputs": [], + "source": [ + "# This will take 5-10 min\n", + "\n", + "recall = sum(\n", + " [\n", + " len(\n", + " set(get_neighbors(test[i], DEPLOYED_BRUTE_FORCE_INDEX_ID)).intersection(\n", + " set(get_neighbors(test[i], DEPLOYED_INDEX_ID))\n", + " )\n", + " )\n", + " for i in range(len(test))\n", + " ]\n", + ") / (1.0 * len(test) * k)\n", + "\n", + "print(\"Recall: {}\".format(recall))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "TpV-iwP9qw9c" + }, + "source": [ + "## Cleaning up\n", + "\n", + "To clean up all Google Cloud resources used in this project, you can [delete the Google Cloud\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "You can also manually delete resources that you created by running the following code." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "sx_vKniMq9ZX" + }, + "outputs": [], + "source": [ + "index_client.delete_index(name=INDEX_RESOURCE_NAME)\n", + "index_client.delete_index(name=INDEX_BRUTE_FORCE_RESOURCE_NAME)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "omj7N9iWv-Tq" + }, + "outputs": [], + "source": [ + "index_endpoint_client.delete_index_endpoint(name=INDEX_ENDPOINT_NAME)" + ] + } + ], + "metadata": { + "colab": { + "collapsed_sections": [], + "name": "matching_engine_for_indexing.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/migration/README.md b/notebooks/community/migration/README.md new file mode 100644 index 000000000..e69de29bb diff --git a/notebooks/community/migration/UJ1 legacy AutoML Vision Image Classification.ipynb b/notebooks/community/migration/UJ1 legacy AutoML Vision Image Classification.ipynb new file mode 100644 index 000000000..c624a629e --- /dev/null +++ b/notebooks/community/migration/UJ1 legacy AutoML Vision Image Classification.ipynb @@ -0,0 +1,3595 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License.\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title:migration,new" + }, + "source": [ + "# AutoML SDK: AutoML image classification model\n", + "\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest (preview) version of AutoML SDK.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "eGO6_2SExXiI", + "outputId": "a966db70-192a-4bb7-f79c-9c9742c09846" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-automl --user\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the Google *cloud-storage* library as well.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "G0bLTzVWxXiJ", + "outputId": "ac5e6d08-d3e4-4076-ac27-6c234b4fc678" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-storage\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the Kernel\n", + "\n", + "Once you've installed the AutoML SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "33Uq-fKkxXiK" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "\n", + "if not os.getenv(\"AUTORUN\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU run-time\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your GCP project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a GCP project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the AutoML APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [Google Cloud SDK](https://cloud.google.com/sdk) is already installed in AutoML Notebooks.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" #@param {type:\"string\"}\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id", + "outputId": "2663b1ca-f6c6-43ed-c23b-654919cb200c" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id", + "outputId": "c401449c-69bb-499a-adc2-ddc15b6cf7f4" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for AutoML. We recommend when possible, to choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You cannot use a Multi-Regional Storage bucket for training with AutoML. Not all regions provide support for all AutoML services. For the latest support per region, see [Region support for AutoML services]()\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "_eH8WcWTxXiM" + }, + "outputs": [], + "source": [ + "REGION = 'us-central1' #@param {type: \"string\"}\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "NoqPBceExXiM" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your GCP account\n", + "\n", + "**If you are using AutoML Notebooks**, your environment is already\n", + "authenticated. Skip this step.\n", + "\n", + "*Note: If you are on an AutoML notebook and run the cell, the cell knows to skip executing the authentication steps.*\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "6aUuFIftxXiN" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your Google Cloud account. This provides access\n", + "# to your Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on AutoML, then don't execute this code\n", + "if not os.path.exists('/opt/deeplearning/metadata/env_version'):\n", + " if 'google.colab' in sys.modules:\n", + " from google.colab import auth as google_auth\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this tutorial in a notebook locally, replace the string\n", + " # below with the path to your service account key and run this cell to\n", + " # authenticate your Google Cloud account.\n", + " else:\n", + " %env GOOGLE_APPLICATION_CREDENTIALS your_path_to_credentials.json\n", + "\n", + " # Log in to your account on Google Cloud\n", + " ! gcloud auth login\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"[your-bucket-name]\" #@param {type:\"string\"}\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n", + " BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "-8qURIGmxXiO", + "outputId": "e887810d-a0fd-4063-d636-20313733b955" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION gs://$BUCKET_NAME\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "t_65JAARxXiP" + }, + "outputs": [], + "source": [ + "! gsutil ls -al gs://$BUCKET_NAME\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip" + }, + "source": [ + "#### Import AutoML SDK\n", + "\n", + "Import the AutoML SDK into our Python environment.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "P8L1tKAexXiQ" + }, + "outputs": [], + "source": [ + "import json\n", + "import os\n", + "import sys\n", + "import time\n", + "\n", + "\n", + "from google.cloud import automl\n", + "\n", + "\n", + "from google.protobuf.json_format import MessageToJson\n", + "from google.protobuf.json_format import ParseDict\n", + "from googleapiclient.discovery import build\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Setup up the following constants for AutoML:\n", + "\n", + "- `PARENT`: The AutoML location root path for dataset, model and endpoint resources.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "rG1L9BEzxXiR" + }, + "outputs": [], + "source": [ + "# AutoML location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients" + }, + "source": [ + "## Clients\n", + "\n", + "The AutoML SDK works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the server (AutoML).\n", + "\n", + "You will use several clients in this tutorial, so set them all up upfront.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "c3sR3OxmxXiR", + "outputId": "5bc213e7-2df8-4aca-dbf8-f8d267e98d52" + }, + "outputs": [], + "source": [ + "def automl_client():\n", + " return automl.AutoMlClient()\n", + "\n", + "\n", + "def prediction_client():\n", + " return automl.PredictionServiceClient()\n", + "\n", + "\n", + "def operations_client():\n", + " return automl.AutoMlClient()._transport.operations_client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"automl\"] = automl_client()\n", + "clients[\"prediction\"] = prediction_client()\n", + "clients[\"operations\"] = operations_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:flowers,csv,icn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://cloud-ml-data/img/flower_photos/train_set.csv\"\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "2do-AnOHxXiS", + "scrolled": true + }, + "outputs": [], + "source": [ + "#%%capture\n", + "! gsutil cp -r gs://cloud-ml-data/img/flower_photos/ gs://$BUCKET_NAME\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "hwMjQtLLxXiS" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "all_files_csv = ! gsutil cat $IMPORT_FILE\n", + "all_files_csv = [ l.replace(\"cloud-ml-data/img\", BUCKET_NAME) for l in all_files_csv ]\n", + "\n", + "IMPORT_FILE = \"gs://\" + BUCKET_NAME + \"/flower_photos/train_set.csv\"\n", + "with tf.io.gfile.GFile(IMPORT_FILE, 'w') as f:\n", + " for l in all_files_csv:\n", + " f.write(l + \"\\n\")\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "! gsutil cat $IMPORT_FILE | head -n 10" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_create:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210226015151/flower_photos/daisy/754296579_30a9ae018c_n.jpg,daisy\n", + "gs://migration-ucaip-trainingaip-20210226015151/flower_photos/dandelion/18089878729_907ed2c7cd_m.jpg,dandelion\n", + "gs://migration-ucaip-trainingaip-20210226015151/flower_photos/dandelion/284497199_93a01f48f6.jpg,dandelion\n", + "gs://migration-ucaip-trainingaip-20210226015151/flower_photos/dandelion/3554992110_81d8c9b0bd_m.jpg,dandelion\n", + "gs://migration-ucaip-trainingaip-20210226015151/flower_photos/daisy/4065883015_4bb6010cb7_n.jpg,daisy\n", + "gs://migration-ucaip-trainingaip-20210226015151/flower_photos/roses/7420699022_60fa574524_m.jpg,roses\n", + "gs://migration-ucaip-trainingaip-20210226015151/flower_photos/dandelion/4558536575_d43a611bd4_n.jpg,dandelion\n", + "gs://migration-ucaip-trainingaip-20210226015151/flower_photos/daisy/7568630428_8cf0fc16ff_n.jpg,daisy\n", + "gs://migration-ucaip-trainingaip-20210226015151/flower_photos/tulips/7064813645_f7f48fb527.jpg,tulips\n", + "gs://migration-ucaip-trainingaip-20210226015151/flower_photos/sunflowers/4933229095_f7e4218b28.jpg,sunflowers\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_a_dataset:migration" + }, + "source": [ + "## Create a dataset\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bkdu67d_xXiT" + }, + "source": [ + "### [projects.locations.datasets.create](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.datasets/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "request:migration" + }, + "source": [ + "#### Request\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "uTrBPMR5xXiU", + "outputId": "8da73a34-a13f-4e1f-c8af-7f25338db142" + }, + "outputs": [], + "source": [ + "dataset = {\n", + " \"display_name\": \"flowers_\" + TIMESTAMP,\n", + " \"image_classification_dataset_metadata\": {\n", + " \"classification_type\": \"MULTICLASS\",\n", + " },\n", + "}\n", + "\n", + "print(MessageToJson(\n", + " automl.CreateDatasetRequest(\n", + " parent=PARENT,\n", + " dataset=dataset,\n", + " ).__dict__[\"_pb\"])\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "jvhVeS_ExXiU" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"dataset\": {\n", + " \"displayName\": \"flowers_20210226015151\",\n", + " \"imageClassificationDatasetMetadata\": {\n", + " \"classificationType\": \"MULTICLASS\"\n", + " }\n", + " }\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "call:migration" + }, + "source": [ + "#### Call\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "datasets_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].create_dataset(\n", + " parent=PARENT,\n", + " dataset=dataset,\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "response:migration" + }, + "source": [ + "#### Response\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,response", + "outputId": "addad607-2542-45c2-d1b1-5ca7aa1d5d0c" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_create:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/datasets/ICN2833688305139187712\"\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:migration,new,response", + "outputId": "d201ed62-717f-4798-9006-a299684d4bf5" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split('/')[-1]\n", + "\n", + "print(dataset_id)\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "qHWi_CMnxXiV" + }, + "source": [ + "### [projects.locations.datasets.importData](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.datasets/importData)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "py-0niVqxXiW" + }, + "source": [ + "#### Request\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "datasets_import:migration,new,request", + "outputId": "83bf0203-7782-42e8-ab78-a23eb1679561" + }, + "outputs": [], + "source": [ + "input_config = {\n", + " \"gcs_source\": {\n", + " \"input_uris\": [IMPORT_FILE],\n", + " },\n", + "}\n", + "\n", + "print(MessageToJson(\n", + " automl.ImportDataRequest(\n", + " name=dataset_short_id,\n", + " input_config=input_config\n", + " ).__dict__[\"_pb\"])\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cxbp1cGzxXiW" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"ICN2833688305139187712\",\n", + " \"inputConfig\": {\n", + " \"gcsSource\": {\n", + " \"inputUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210226015151/flower_photos/train_set.csv\"\n", + " ]\n", + " }\n", + " }\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "v8kGJFwYxXiW" + }, + "source": [ + "#### Call\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "datasets_import:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].import_data(\n", + " name=dataset_id,\n", + " input_config=input_config\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "UxcyIvGLxXiW" + }, + "source": [ + "#### Response\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "zXlYWLR7xXiX", + "outputId": "d5740c78-a8db-4d69-84d7-72cf3b83cf87" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_a_model:migration" + }, + "source": [ + "## Train a model\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_models_create:migration,old" + }, + "source": [ + "### [projects.locations.models.create](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bytx9mM0xXiX" + }, + "source": [ + "#### Request\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_create:migration,new,request,icn", + "outputId": "4156101b-6aa6-48d3-d4de-277ab6c3757e" + }, + "outputs": [], + "source": [ + "model = {\n", + " \"display_name\": \"flowers_\" + TIMESTAMP,\n", + " \"dataset_id\": dataset_short_id,\n", + " \"image_classification_model_metadata\": {\n", + " \"train_budget_milli_node_hours\": 8000,\n", + " },\n", + "}\n", + "\n", + "print(MessageToJson(\n", + " automl.CreateModelRequest(\n", + " parent=PARENT,\n", + " model=model,\n", + " ).__dict__[\"_pb\"])\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wF6C0IJcxXiY" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"model\": {\n", + " \"displayName\": \"flowers_20210226015151\",\n", + " \"datasetId\": \"ICN2833688305139187712\",\n", + " \"imageClassificationModelMetadata\": {\n", + " \"trainBudgetMilliNodeHours\": \"8000\"\n", + " }\n", + " }\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cNNI6bjcxXiY" + }, + "source": [ + "#### Call\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].create_model(\n", + " parent=PARENT,\n", + " model=model,\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ZBPeFwbQxXiY" + }, + "source": [ + "#### Response\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,request", + "outputId": "8f64b370-c390-4f6d-8f9f-d8204a49bcda" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "trainingpipelines_create:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/ICN3600040762873479168\"\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "training_pipeline_id:migration,new,response", + "outputId": "484821ac-6ab3-4a26-c3af-b2008661606d" + }, + "outputs": [], + "source": [ + "# The full unique ID for the training pipeline\n", + "model_id = result.name\n", + "# The short numeric ID for the training pipeline\n", + "model_short_id = model_id.split('/')[-1]\n", + "\n", + "print(model_short_id)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:migration" + }, + "source": [ + "## Evaluate the model\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "yfufMwAEJjkX" + }, + "source": [ + "### [projects.locations.models.modelEvaluations.list](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models.modelEvaluations/list)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "LA6VndrpxXiZ" + }, + "source": [ + "#### Call\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "models_evaluations_list:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].list_model_evaluations(\n", + " parent=model_id, \n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "8CyKSW9wxXiZ" + }, + "source": [ + "#### Response\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "AxEY7WFHj2YC", + "outputId": "e284ae12-dbb8-4b8f-c576-03d23cc822af" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "\n", + "model_evaluations = [\n", + " json.loads(MessageToJson(me.__dict__[\"_pb\"])) for me in request \n", + "]\n", + "# The evaluation slice\n", + "evaluation_slice = request.model_evaluation[0].name\n", + "\n", + "print(json.dumps(model_evaluations, indent=2))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "models_evaluations_list:migration,new,response,icn" + }, + "source": [ + "*Example output*\n", + "```\n", + "[\n", + " {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/ICN3600040762873479168/modelEvaluations/1701367336556072668\",\n", + " \"createTime\": \"2021-02-26T03:00:19.383521Z\",\n", + " \"evaluatedExampleCount\": 329,\n", + " \"classificationEvaluationMetrics\": {\n", + " \"auPrc\": 0.99747145,\n", + " \"confidenceMetricsEntry\": [\n", + " {\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.2\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.05,\n", + " \"recall\": 0.99088144,\n", + " \"precision\": 0.92877495\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.1,\n", + " \"recall\": 0.98784196,\n", + " \"precision\": 0.9447674\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.15,\n", + " \"recall\": 0.9848024,\n", + " \"precision\": 0.9501466\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.2,\n", + " \"recall\": 0.9848024,\n", + " \"precision\": 0.96142435\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.25,\n", + " \"recall\": 0.98176295,\n", + " \"precision\": 0.9641791\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.3,\n", + " \"recall\": 0.98176295,\n", + " \"precision\": 0.9670659\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.35,\n", + " \"recall\": 0.9787234,\n", + " \"precision\": 0.966967\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.4,\n", + " \"recall\": 0.97568387,\n", + " \"precision\": 0.96686745\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.45,\n", + " \"recall\": 0.97568387,\n", + " \"precision\": 0.9727273\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5,\n", + " \"recall\": 0.9726444,\n", + " \"precision\": 0.9756098\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.55,\n", + " \"recall\": 0.9726444,\n", + " \"precision\": 0.9756098\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6,\n", + " \"recall\": 0.9665654,\n", + " \"precision\": 0.9754601\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.65,\n", + " \"recall\": 0.9665654,\n", + " \"precision\": 0.9814815\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.7,\n", + " \"recall\": 0.9665654,\n", + " \"precision\": 0.98452014\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.75,\n", + " \"recall\": 0.9665654,\n", + " \"precision\": 0.98452014\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.8,\n", + " \"recall\": 0.9604863,\n", + " \"precision\": 0.9875\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.85,\n", + " \"recall\": 0.9452888,\n", + " \"precision\": 0.99044585\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.875,\n", + " \"recall\": 0.94224924,\n", + " \"precision\": 0.99041533\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9,\n", + " \"recall\": 0.9392097,\n", + " \"precision\": 0.99038464\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.91,\n", + " \"recall\": 0.9392097,\n", + " \"precision\": 0.99038464\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.92,\n", + " \"recall\": 0.9361702,\n", + " \"precision\": 0.9935484\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.93,\n", + " \"recall\": 0.9361702,\n", + " \"precision\": 0.9935484\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.94,\n", + " \"recall\": 0.9361702,\n", + " \"precision\": 0.9935484\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.95,\n", + " \"recall\": 0.9331307,\n", + " \"precision\": 0.99352753\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.96,\n", + " \"recall\": 0.9300912,\n", + " \"precision\": 0.99674267\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.97,\n", + " \"recall\": 0.92705166,\n", + " \"precision\": 0.996732\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98,\n", + " \"recall\": 0.9148936,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99,\n", + " \"recall\": 0.89361703,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.995,\n", + " \"recall\": 0.88145894,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.996,\n", + " \"recall\": 0.87234044,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.997,\n", + " \"recall\": 0.8693009,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.998,\n", + " \"recall\": 0.8449848,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.999,\n", + " \"recall\": 0.81155014,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 1.0,\n", + " \"recall\": 0.24012157,\n", + " \"precision\": 1.0\n", + " }\n", + " ],\n", + " \"confusionMatrix\": {\n", + " \"annotationSpecId\": [\n", + " \"548545251585818624\",\n", + " \"4295540141558071296\",\n", + " \"5160231270013206528\",\n", + " \"6601383150771765248\",\n", + " \"8907226159985459200\"\n", + " ],\n", + " \"row\": [\n", + " {\n", + " \"exampleCount\": [\n", + " 55,\n", + " 0,\n", + " 1,\n", + " 2,\n", + " 0\n", + " ]\n", + " },\n", + " {\n", + " \"exampleCount\": [\n", + " 0,\n", + " 59,\n", + " 1,\n", + " 0,\n", + " 1\n", + " ]\n", + " },\n", + " {\n", + " \"exampleCount\": [\n", + " 0,\n", + " 0,\n", + " 81,\n", + " 0,\n", + " 0\n", + " ]\n", + " },\n", + " {\n", + " \"exampleCount\": [\n", + " 0,\n", + " 0,\n", + " 0,\n", + " 73,\n", + " 0\n", + " ]\n", + " },\n", + " {\n", + " \"exampleCount\": [\n", + " 0,\n", + " 1,\n", + " 2,\n", + " 0,\n", + " 53\n", + " ]\n", + " }\n", + " ],\n", + " \"displayName\": [\n", + " \"roses\",\n", + " \"sunflowers\",\n", + " \"dandelion\",\n", + " \"tulips\",\n", + " \"daisy\"\n", + " ]\n", + " },\n", + " \"logLoss\": 0.02853713\n", + " }\n", + " },\n", + " {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/ICN3600040762873479168/modelEvaluations/4464795143994212237\",\n", + " \"annotationSpecId\": \"6601383150771765248\",\n", + " \"createTime\": \"2021-02-26T03:00:19.383521Z\",\n", + " \"classificationEvaluationMetrics\": {\n", + " \"auPrc\": 0.9990742,\n", + " \"confidenceMetricsEntry\": [\n", + " {\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.2218845\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.05,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.8795181\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.1,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.9125\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.15,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.9240506\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.2,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.9605263\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.25,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.9605263\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.3,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.97333336\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.35,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.97333336\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.4,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.97333336\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.45,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.97333336\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5,\n", + " \"recall\": 0.98630136,\n", + " \"precision\": 0.972973\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.55,\n", + " \"recall\": 0.98630136,\n", + " \"precision\": 0.972973\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6,\n", + " \"recall\": 0.9726027,\n", + " \"precision\": 0.9726027\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.65,\n", + " \"recall\": 0.9726027,\n", + " \"precision\": 0.9726027\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.7,\n", + " \"recall\": 0.9726027,\n", + " \"precision\": 0.9726027\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.75,\n", + " \"recall\": 0.9726027,\n", + " \"precision\": 0.9726027\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.8,\n", + " \"recall\": 0.9726027,\n", + " \"precision\": 0.9861111\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.85,\n", + " \"recall\": 0.9726027,\n", + " \"precision\": 0.9861111\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.875,\n", + " \"recall\": 0.9726027,\n", + " \"precision\": 0.9861111\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9,\n", + " \"recall\": 0.9726027,\n", + " \"precision\": 0.9861111\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.91,\n", + " \"recall\": 0.9726027,\n", + " \"precision\": 0.9861111\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.92,\n", + " \"recall\": 0.9726027,\n", + " \"precision\": 0.9861111\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.93,\n", + " \"recall\": 0.9726027,\n", + " \"precision\": 0.9861111\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.94,\n", + " \"recall\": 0.9726027,\n", + " \"precision\": 0.9861111\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.95,\n", + " \"recall\": 0.9726027,\n", + " \"precision\": 0.9861111\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.96,\n", + " \"recall\": 0.9726027,\n", + " \"precision\": 0.9861111\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.97,\n", + " \"recall\": 0.9726027,\n", + " \"precision\": 0.9861111\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98,\n", + " \"recall\": 0.9589041,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99,\n", + " \"recall\": 0.9315069,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.995,\n", + " \"recall\": 0.91780823,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.996,\n", + " \"recall\": 0.91780823,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.997,\n", + " \"recall\": 0.91780823,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.998,\n", + " \"recall\": 0.9041096,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.999,\n", + " \"recall\": 0.8356164,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 1.0,\n", + " \"recall\": 0.12328767,\n", + " \"precision\": 1.0\n", + " }\n", + " ],\n", + " \"auRoc\": 0.99973243,\n", + " \"logLoss\": 0.024023052\n", + " },\n", + " \"displayName\": \"tulips\"\n", + " },\n", + " {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/ICN3600040762873479168/modelEvaluations/6132683338167493052\",\n", + " \"annotationSpecId\": \"8907226159985459200\",\n", + " \"createTime\": \"2021-02-26T03:00:19.383521Z\",\n", + " \"classificationEvaluationMetrics\": {\n", + " \"auPrc\": 0.99841,\n", + " \"confidenceMetricsEntry\": [\n", + " {\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.17021276\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.05,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.9655172\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.1,\n", + " \"recall\": 0.98214287,\n", + " \"precision\": 0.9649123\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.15,\n", + " \"recall\": 0.98214287,\n", + " \"precision\": 0.98214287\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.2,\n", + " \"recall\": 0.98214287,\n", + " \"precision\": 0.98214287\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.25,\n", + " \"recall\": 0.98214287,\n", + " \"precision\": 0.98214287\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.3,\n", + " \"recall\": 0.98214287,\n", + " \"precision\": 0.98214287\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.35,\n", + " \"recall\": 0.96428573,\n", + " \"precision\": 0.9818182\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.4,\n", + " \"recall\": 0.9464286,\n", + " \"precision\": 0.9814815\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.45,\n", + " \"recall\": 0.9464286,\n", + " \"precision\": 0.9814815\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5,\n", + " \"recall\": 0.9464286,\n", + " \"precision\": 0.9814815\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.55,\n", + " \"recall\": 0.9464286,\n", + " \"precision\": 0.9814815\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6,\n", + " \"recall\": 0.9285714,\n", + " \"precision\": 0.9811321\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.65,\n", + " \"recall\": 0.9285714,\n", + " \"precision\": 0.9811321\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.7,\n", + " \"recall\": 0.9285714,\n", + " \"precision\": 0.9811321\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.75,\n", + " \"recall\": 0.9285714,\n", + " \"precision\": 0.9811321\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.8,\n", + " \"recall\": 0.9285714,\n", + " \"precision\": 0.9811321\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.85,\n", + " \"recall\": 0.9285714,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.875,\n", + " \"recall\": 0.9285714,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9,\n", + " \"recall\": 0.9285714,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.91,\n", + " \"recall\": 0.9285714,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.92,\n", + " \"recall\": 0.9285714,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.93,\n", + " \"recall\": 0.9285714,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.94,\n", + " \"recall\": 0.9285714,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.95,\n", + " \"recall\": 0.91071427,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.96,\n", + " \"recall\": 0.91071427,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.97,\n", + " \"recall\": 0.91071427,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98,\n", + " \"recall\": 0.91071427,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99,\n", + " \"recall\": 0.875,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.995,\n", + " \"recall\": 0.83928573,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.996,\n", + " \"recall\": 0.8214286,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.997,\n", + " \"recall\": 0.8035714,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.998,\n", + " \"recall\": 0.78571427,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.999,\n", + " \"recall\": 0.76785713,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 1.0,\n", + " \"recall\": 0.30357143,\n", + " \"precision\": 1.0\n", + " }\n", + " ],\n", + " \"auRoc\": 0.99967295,\n", + " \"logLoss\": 0.022124559\n", + " },\n", + " \"displayName\": \"daisy\"\n", + " },\n", + " {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/ICN3600040762873479168/modelEvaluations/7147485663377408481\",\n", + " \"annotationSpecId\": \"548545251585818624\",\n", + " \"createTime\": \"2021-02-26T03:00:19.383521Z\",\n", + " \"classificationEvaluationMetrics\": {\n", + " \"auPrc\": 0.9971625,\n", + " \"confidenceMetricsEntry\": [\n", + " {\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.1762918\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.05,\n", + " \"recall\": 0.9655172,\n", + " \"precision\": 0.93333334\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.1,\n", + " \"recall\": 0.9655172,\n", + " \"precision\": 0.9655172\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.15,\n", + " \"recall\": 0.9655172,\n", + " \"precision\": 0.9655172\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.2,\n", + " \"recall\": 0.9655172,\n", + " \"precision\": 0.9655172\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.25,\n", + " \"recall\": 0.94827586,\n", + " \"precision\": 0.9649123\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.3,\n", + " \"recall\": 0.94827586,\n", + " \"precision\": 0.9649123\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.35,\n", + " \"recall\": 0.94827586,\n", + " \"precision\": 0.9649123\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.4,\n", + " \"recall\": 0.94827586,\n", + " \"precision\": 0.9649123\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.45,\n", + " \"recall\": 0.94827586,\n", + " \"precision\": 0.98214287\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5,\n", + " \"recall\": 0.94827586,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.55,\n", + " \"recall\": 0.94827586,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6,\n", + " \"recall\": 0.94827586,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.65,\n", + " \"recall\": 0.94827586,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.7,\n", + " \"recall\": 0.94827586,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.75,\n", + " \"recall\": 0.94827586,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.8,\n", + " \"recall\": 0.94827586,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.85,\n", + " \"recall\": 0.87931037,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.875,\n", + " \"recall\": 0.87931037,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9,\n", + " \"recall\": 0.87931037,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.91,\n", + " \"recall\": 0.87931037,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.92,\n", + " \"recall\": 0.86206895,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.93,\n", + " \"recall\": 0.86206895,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.94,\n", + " \"recall\": 0.86206895,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.95,\n", + " \"recall\": 0.86206895,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.96,\n", + " \"recall\": 0.86206895,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.97,\n", + " \"recall\": 0.86206895,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98,\n", + " \"recall\": 0.8448276,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99,\n", + " \"recall\": 0.79310346,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.995,\n", + " \"recall\": 0.79310346,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.996,\n", + " \"recall\": 0.7758621,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.997,\n", + " \"recall\": 0.7758621,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.998,\n", + " \"recall\": 0.70689654,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.999,\n", + " \"recall\": 0.6896552,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 1.0,\n", + " \"recall\": 0.03448276,\n", + " \"precision\": 1.0\n", + " }\n", + " ],\n", + " \"auRoc\": 0.9993638,\n", + " \"logLoss\": 0.034111425\n", + " },\n", + " \"displayName\": \"roses\"\n", + " },\n", + " {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/ICN3600040762873479168/modelEvaluations/8076647367053688867\",\n", + " \"annotationSpecId\": \"5160231270013206528\",\n", + " \"createTime\": \"2021-02-26T03:00:19.383521Z\",\n", + " \"classificationEvaluationMetrics\": {\n", + " \"auPrc\": 0.9989403,\n", + " \"confidenceMetricsEntry\": [\n", + " {\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.2462006\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.05,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.9101124\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.1,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.92045456\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.15,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.92045456\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.2,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.9310345\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.25,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.94186044\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.3,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.94186044\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.35,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.94186044\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.4,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.94186044\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.45,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.9529412\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.9529412\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.55,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.9529412\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.9529412\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.65,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.96428573\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.7,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.97590363\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.75,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.97590363\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.8,\n", + " \"recall\": 0.9876543,\n", + " \"precision\": 0.9756098\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.85,\n", + " \"recall\": 0.9876543,\n", + " \"precision\": 0.9756098\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.875,\n", + " \"recall\": 0.97530866,\n", + " \"precision\": 0.97530866\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9,\n", + " \"recall\": 0.962963,\n", + " \"precision\": 0.975\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.91,\n", + " \"recall\": 0.962963,\n", + " \"precision\": 0.975\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.92,\n", + " \"recall\": 0.962963,\n", + " \"precision\": 0.98734176\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.93,\n", + " \"recall\": 0.962963,\n", + " \"precision\": 0.98734176\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.94,\n", + " \"recall\": 0.962963,\n", + " \"precision\": 0.98734176\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.95,\n", + " \"recall\": 0.962963,\n", + " \"precision\": 0.98734176\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.96,\n", + " \"recall\": 0.9506173,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.97,\n", + " \"recall\": 0.9506173,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98,\n", + " \"recall\": 0.9506173,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99,\n", + " \"recall\": 0.9506173,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.995,\n", + " \"recall\": 0.9506173,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.996,\n", + " \"recall\": 0.9506173,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.997,\n", + " \"recall\": 0.9506173,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.998,\n", + " \"recall\": 0.9382716,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.999,\n", + " \"recall\": 0.9259259,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 1.0,\n", + " \"recall\": 0.5555556,\n", + " \"precision\": 1.0\n", + " }\n", + " ],\n", + " \"auRoc\": 0.99965155,\n", + " \"logLoss\": 0.029262401\n", + " },\n", + " \"displayName\": \"dandelion\"\n", + " },\n", + " {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/ICN3600040762873479168/modelEvaluations/8816571236383372686\",\n", + " \"annotationSpecId\": \"4295540141558071296\",\n", + " \"createTime\": \"2021-02-26T03:00:19.383521Z\",\n", + " \"classificationEvaluationMetrics\": {\n", + " \"auPrc\": 0.99703646,\n", + " \"confidenceMetricsEntry\": [\n", + " {\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.18541034\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.05,\n", + " \"recall\": 0.9836066,\n", + " \"precision\": 0.9836066\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.1,\n", + " \"recall\": 0.9836066,\n", + " \"precision\": 0.9836066\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.15,\n", + " \"recall\": 0.9672131,\n", + " \"precision\": 0.98333335\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.2,\n", + " \"recall\": 0.9672131,\n", + " \"precision\": 0.98333335\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.25,\n", + " \"recall\": 0.9672131,\n", + " \"precision\": 0.98333335\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.3,\n", + " \"recall\": 0.9672131,\n", + " \"precision\": 0.98333335\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.35,\n", + " \"recall\": 0.9672131,\n", + " \"precision\": 0.98333335\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.4,\n", + " \"recall\": 0.9672131,\n", + " \"precision\": 0.98333335\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.45,\n", + " \"recall\": 0.9672131,\n", + " \"precision\": 0.98333335\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5,\n", + " \"recall\": 0.9672131,\n", + " \"precision\": 0.98333335\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.55,\n", + " \"recall\": 0.9672131,\n", + " \"precision\": 0.98333335\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6,\n", + " \"recall\": 0.9672131,\n", + " \"precision\": 0.98333335\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.65,\n", + " \"recall\": 0.9672131,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.7,\n", + " \"recall\": 0.9672131,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.75,\n", + " \"recall\": 0.9672131,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.8,\n", + " \"recall\": 0.9508197,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.85,\n", + " \"recall\": 0.93442625,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.875,\n", + " \"recall\": 0.93442625,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9,\n", + " \"recall\": 0.93442625,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.91,\n", + " \"recall\": 0.93442625,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.92,\n", + " \"recall\": 0.93442625,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.93,\n", + " \"recall\": 0.93442625,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.94,\n", + " \"recall\": 0.93442625,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.95,\n", + " \"recall\": 0.93442625,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.96,\n", + " \"recall\": 0.93442625,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.97,\n", + " \"recall\": 0.91803277,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98,\n", + " \"recall\": 0.8852459,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99,\n", + " \"recall\": 0.8852459,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.995,\n", + " \"recall\": 0.86885244,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.996,\n", + " \"recall\": 0.852459,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.997,\n", + " \"recall\": 0.852459,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.998,\n", + " \"recall\": 0.8360656,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.999,\n", + " \"recall\": 0.78688526,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 1.0,\n", + " \"recall\": 0.09836066,\n", + " \"precision\": 1.0\n", + " }\n", + " ],\n", + " \"auRoc\": 0.9992048,\n", + " \"logLoss\": 0.03316421\n", + " },\n", + " \"displayName\": \"sunflowers\"\n", + " }\n", + "]\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "i6T0bzuNJjkY" + }, + "source": [ + "### [projects.locations.models.modelEvaluations.get](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models.modelEvaluations/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "KJ1zoYmrxXia" + }, + "source": [ + "#### Call\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "models_evaluations_get:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].get_model_evaluation(\n", + " name=evaluation_slice,\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "_uZksIkGxXia" + }, + "source": [ + "#### Response\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cEqg6FHvxXib", + "outputId": "135ca63e-f2ba-44d5-8a2a-df478d2335e3", + "scrolled": true + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "models_evaluations_get:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/ICN3600040762873479168/modelEvaluations/1701367336556072668\",\n", + " \"createTime\": \"2021-02-26T03:00:19.383521Z\",\n", + " \"evaluatedExampleCount\": 329,\n", + " \"classificationEvaluationMetrics\": {\n", + " \"auPrc\": 0.99747145,\n", + " \"confidenceMetricsEntry\": [\n", + " {\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.2\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.05,\n", + " \"recall\": 0.99088144,\n", + " \"precision\": 0.92877495\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.1,\n", + " \"recall\": 0.98784196,\n", + " \"precision\": 0.9447674\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.15,\n", + " \"recall\": 0.9848024,\n", + " \"precision\": 0.9501466\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.2,\n", + " \"recall\": 0.9848024,\n", + " \"precision\": 0.96142435\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.25,\n", + " \"recall\": 0.98176295,\n", + " \"precision\": 0.9641791\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.3,\n", + " \"recall\": 0.98176295,\n", + " \"precision\": 0.9670659\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.35,\n", + " \"recall\": 0.9787234,\n", + " \"precision\": 0.966967\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.4,\n", + " \"recall\": 0.97568387,\n", + " \"precision\": 0.96686745\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.45,\n", + " \"recall\": 0.97568387,\n", + " \"precision\": 0.9727273\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5,\n", + " \"recall\": 0.9726444,\n", + " \"precision\": 0.9756098\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.55,\n", + " \"recall\": 0.9726444,\n", + " \"precision\": 0.9756098\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6,\n", + " \"recall\": 0.9665654,\n", + " \"precision\": 0.9754601\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.65,\n", + " \"recall\": 0.9665654,\n", + " \"precision\": 0.9814815\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.7,\n", + " \"recall\": 0.9665654,\n", + " \"precision\": 0.98452014\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.75,\n", + " \"recall\": 0.9665654,\n", + " \"precision\": 0.98452014\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.8,\n", + " \"recall\": 0.9604863,\n", + " \"precision\": 0.9875\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.85,\n", + " \"recall\": 0.9452888,\n", + " \"precision\": 0.99044585\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.875,\n", + " \"recall\": 0.94224924,\n", + " \"precision\": 0.99041533\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9,\n", + " \"recall\": 0.9392097,\n", + " \"precision\": 0.99038464\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.91,\n", + " \"recall\": 0.9392097,\n", + " \"precision\": 0.99038464\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.92,\n", + " \"recall\": 0.9361702,\n", + " \"precision\": 0.9935484\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.93,\n", + " \"recall\": 0.9361702,\n", + " \"precision\": 0.9935484\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.94,\n", + " \"recall\": 0.9361702,\n", + " \"precision\": 0.9935484\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.95,\n", + " \"recall\": 0.9331307,\n", + " \"precision\": 0.99352753\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.96,\n", + " \"recall\": 0.9300912,\n", + " \"precision\": 0.99674267\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.97,\n", + " \"recall\": 0.92705166,\n", + " \"precision\": 0.996732\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98,\n", + " \"recall\": 0.9148936,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99,\n", + " \"recall\": 0.89361703,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.995,\n", + " \"recall\": 0.88145894,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.996,\n", + " \"recall\": 0.87234044,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.997,\n", + " \"recall\": 0.8693009,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.998,\n", + " \"recall\": 0.8449848,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.999,\n", + " \"recall\": 0.81155014,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 1.0,\n", + " \"recall\": 0.24012157,\n", + " \"precision\": 1.0\n", + " }\n", + " ],\n", + " \"confusionMatrix\": {\n", + " \"annotationSpecId\": [\n", + " \"548545251585818624\",\n", + " \"4295540141558071296\",\n", + " \"5160231270013206528\",\n", + " \"6601383150771765248\",\n", + " \"8907226159985459200\"\n", + " ],\n", + " \"row\": [\n", + " {\n", + " \"exampleCount\": [\n", + " 55,\n", + " 0,\n", + " 1,\n", + " 2,\n", + " 0\n", + " ]\n", + " },\n", + " {\n", + " \"exampleCount\": [\n", + " 0,\n", + " 59,\n", + " 1,\n", + " 0,\n", + " 1\n", + " ]\n", + " },\n", + " {\n", + " \"exampleCount\": [\n", + " 0,\n", + " 0,\n", + " 81,\n", + " 0,\n", + " 0\n", + " ]\n", + " },\n", + " {\n", + " \"exampleCount\": [\n", + " 0,\n", + " 0,\n", + " 0,\n", + " 73,\n", + " 0\n", + " ]\n", + " },\n", + " {\n", + " \"exampleCount\": [\n", + " 0,\n", + " 1,\n", + " 2,\n", + " 0,\n", + " 53\n", + " ]\n", + " }\n", + " ],\n", + " \"displayName\": [\n", + " \"roses\",\n", + " \"sunflowers\",\n", + " \"dandelion\",\n", + " \"tulips\",\n", + " \"daisy\"\n", + " ]\n", + " },\n", + " \"logLoss\": 0.02853713\n", + " }\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_predictions:migration" + }, + "source": [ + "## Make batch predictions\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_prediction_file:migration,new" + }, + "source": [ + "### Make a batch prediction file\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_items:automl,icn,csv", + "outputId": "ccbb9682-59dd-464e-96a4-af102b28113e" + }, + "outputs": [], + "source": [ + "test_items = !gsutil cat $IMPORT_FILE | head -n2\n", + "\n", + "if len(str(test_items[0]).split(',')) == 3:\n", + " _, test_item_1, test_label_1 = str(test_items[0]).split(',')\n", + " _, test_item_2, test_label_2 = str(test_items[1]).split(',')\n", + "else:\n", + " test_item_1, test_label_1 = str(test_items[0]).split(',')\n", + " test_item_2, test_label_2 = str(test_items[1]).split(',')\n", + "\n", + "print(test_item_1, test_label_1)\n", + "print(test_item_2, test_label_2)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "fPuu9othxXib" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210226015151/flower_photos/daisy/754296579_30a9ae018c_n.jpg daisy\n", + "gs://migration-ucaip-trainingaip-20210226015151/flower_photos/dandelion/18089878729_907ed2c7cd_m.jpg dandelion\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copy_test_items:automl,image,batch_prediction", + "outputId": "6d4ff7d7-5213-4e78-82e4-c559f55af6ec" + }, + "outputs": [], + "source": [ + "file_1 = test_item_1.split('/')[-1]\n", + "file_2 = test_item_2.split('/')[-1]\n", + "\n", + "! gsutil cp $test_item_1 gs://$BUCKET_NAME/$file_1\n", + "! gsutil cp $test_item_2 gs://$BUCKET_NAME/$file_2\n", + "\n", + "test_item_1 = \"gs://\" + BUCKET_NAME + \"/\" + file_1\n", + "test_item_2 = \"gs://\" + BUCKET_NAME + \"/\" + file_2\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:automl,image" + }, + "source": [ + "### Make the batch input file\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "SY29TGSVxXic", + "outputId": "15e99cb6-cf3e-4133-d5cf-815a0c6b39f4" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "import json\n", + "\n", + "\n", + "gcs_input_uri = \"gs://\" + BUCKET_NAME + '/test.csv'\n", + "with tf.io.gfile.GFile(gcs_input_uri, 'w') as f:\n", + " f.write(test_item_1 + '\\n')\n", + " f.write(test_item_2 + '\\n')\n", + "\n", + "!gsutil cat $gcs_input_uri\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Bc2Hnu9-xXic" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210226015151/754296579_30a9ae018c_n.jpg\n", + "gs://migration-ucaip-trainingaip-20210226015151/18089878729_907ed2c7cd_m.jpg\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_models_batchpredict:migration,old" + }, + "source": [ + "### [projects.locations.models.batchPredict](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models/batchPredict)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "6udI5yImxXid" + }, + "source": [ + "#### Request\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_create:migration,new,request,icn", + "outputId": "1c80667a-65a1-4b53-b2ef-3a1d3213453d" + }, + "outputs": [], + "source": [ + "input_config = {\n", + " \"gcs_source\": {\n", + " \"input_uris\": [gcs_input_uri]\n", + " },\n", + "}\n", + "\n", + "output_config = {\n", + " \"gcs_destination\": {\n", + " \"output_uri_prefix\": \"gs://\" + f\"{BUCKET_NAME}/batch_output/\"\n", + " }\n", + "}\n", + "\n", + "batch_prediction = automl.BatchPredictRequest(\n", + " name=model_id,\n", + " input_config=input_config,\n", + " output_config=output_config\n", + ")\n", + "\n", + "print(MessageToJson(\n", + " batch_prediction.__dict__[\"_pb\"])\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "u10t0ejsxXid" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/ICN3600040762873479168\",\n", + " \"inputConfig\": {\n", + " \"gcsSource\": {\n", + " \"inputUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210226015151/test.csv\"\n", + " ]\n", + " }\n", + " },\n", + " \"outputConfig\": {\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210226015151/batch_output/\"\n", + " }\n", + " }\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "hax200orxXid" + }, + "source": [ + "#### Call\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"prediction\"].batch_predict(\n", + " request=batch_prediction\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "DzaQAgVCxXid" + }, + "source": [ + "#### Response\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "GwBukgk1xXie", + "outputId": "4b0fabcb-9025-4824-9f74-cff6b0a1d856" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_create:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "fnYRCde3xXie", + "outputId": "7800b730-ccfb-42f6-b8a9-6a3531e830c7" + }, + "outputs": [], + "source": [ + "destination_uri = batch_prediction.output_config.gcs_destination.output_uri_prefix[:-1]\n", + "\n", + "! gsutil ls $destination_uri/*\n", + "! gsutil cat $destination_uri/prediction*/*.jsonl\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gbjPT-mLxXie" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210226015151/batch_output/prediction-flowers_20210226015151-2021-02-26T03:00:47.533913Z/image_classification_0.jsonl\n", + "gs://migration-ucaip-trainingaip-20210226015151/batch_output/prediction-flowers_20210226015151-2021-02-26T03:00:47.533913Z/image_classification_1.jsonl\n", + "{\"ID\":\"gs://migration-ucaip-trainingaip-20210226015151/18089878729_907ed2c7cd_m.jpg\",\"annotations\":[{\"annotation_spec_id\":\"5160231270013206528\",\"classification\":{\"score\":0.9993481},\"display_name\":\"dandelion\"}]}\n", + "{\"ID\":\"gs://migration-ucaip-trainingaip-20210226015151/754296579_30a9ae018c_n.jpg\",\"annotations\":[{\"annotation_spec_id\":\"8907226159985459200\",\"classification\":{\"score\":1},\"display_name\":\"daisy\"}]}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "lA-sKaK8xXie" + }, + "source": [ + "## Make online predictions" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "5j4eOtdHxXif" + }, + "source": [ + "### Prepare file for online prediction" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_models_deploy:migration,old" + }, + "source": [ + "### [projects.locations.models.deploy](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models/deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Vo3Dq7IQxXif" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "QVIojpOdxXif" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].deploy_model(\n", + " name=model_id\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "VrWzqRt6xXif" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "86D1mJOoxXif", + "outputId": "cba0d7f8-e84f-432c-810c-00d2ce680c3a" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bB3px5SBxXig" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_models_predict:migration,old" + }, + "source": [ + "### [projects.locations.models.predict](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models/predict)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "LRgEK5aaxXig" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "UOwIprmDxXig", + "outputId": "c90776f2-2084-44e7-af61-9531cd8c58d4", + "scrolled": true + }, + "outputs": [], + "source": [ + "test_item = !gsutil cat $IMPORT_FILE | head -n1\n", + "test_item = test_item[0].split(\",\")[0]\n", + "\n", + "with tf.io.gfile.GFile(test_item, \"rb\") as f:\n", + " content = f.read()\n", + "\n", + "payload = [{\n", + " \"image\": {\n", + " \"image_bytes\": content\n", + " }\n", + "}]\n", + "\n", + "params = {\"score_threshold\": \"0.8\"}\n", + "\n", + "prediction_r = automl.PredictRequest(\n", + " name=model_id,\n", + " payload=payload,\n", + " params=params\n", + ")\n", + "\n", + "print(MessageToJson(\n", + " automl.PredictRequest(\n", + " name=model_id,\n", + " payload=payload,\n", + " params=params\n", + " ).__dict__[\"_pb\"])\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "_3xVOX8sxXih" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/ICN3600040762873479168\",\n", + " \"payload\": {\n", + " \"image\": {\n", + " \"imageBytes\": \"/9j/4AAQSkZJRgABAQAAAQABAAD/4gRISUNDX1BST0ZJTEUAAQEAAAQ4YXBwbAIgAABtbnRyUkdCIFhZWiAH0AAIAA0AEAAGAAdhY3NwQVBQTAAAAABhcHBsAAAAAAAAAAAAAAAAAAAAAQAA9tYAAQAAAADTLWFwcGwAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAAxjcHJ0AAACBAAAAEhkZXNjAAABFAAAADF3dHB0AAABSAAAABRyVFJDAAABXAAAAA5nVFJDAAABXAAAAA5iVFJDAAABXAAAAA5yWFlaAAABbAAAABRnWFlaAAABgAAAABRiWFlaAAABlAAAABR2Y2d0AAABqAAAADBjaGFkAAAB2AAAACxkc2NtAAACTAAAAepkZXNjAAAAAAAAAA1zUkdCIFByb2ZpbGUAAAAAAAAAAAAAAA1zUkdCIFByb2ZpbGUAAAAAWFlaIAAAAAAAAPNRAAEAAAABFsxjdXJ2AAAAAAAAAAECMwAAWFlaIAAAAAAAAG+iAAA49QAAA5BYWVogAAAAAAAAYpkAALeFAAAY2lhZWiAAAAAAAAAkoAAAD4QAALbPdmNndAAAAAAAAAABAADhSAAAAAAAAQAAAADhSAAAAAAAAQAAAADhSAAAAAAAAQAAc2YzMgAAAAAAAQxCAAAF3v//8yYAAAeTAAD9kP//+6L///2jAAAD3AAAwG50ZXh0AAAAAENvcHlyaWdodCAxOTk4IC0gMjAwMyBBcHBsZSBDb21wdXRlciBJbmMuLCBhbGwgcmlnaHRzIHJlc2VydmVkLgBtbHVjAAAAAAAAAA8AAAAMZW5VUwAAABgAAAHSZXNFUwAAABYAAAEyZGFESwAAACAAAAFwZGVERQAAABYAAAFIZmlGSQAAABoAAADEZnJGVQAAABYAAAD0aXRJVAAAABgAAAG6bmxOTAAAABgAAAGQbm9OTwAAABYAAADecHRCUgAAABYAAAEyc3ZTRQAAABYAAADeamFKUAAAABYAAAEKa29LUgAAABIAAAGoemhUVwAAABIAAAEgemhDTgAAABIAAAFeAHMAUgBHAEIALQBwAHIAbwBmAGkAaQBsAGkAcwBSAEcAQgAtAHAAcgBvAGYAaQBsAFAAcgBvAGYAaQBsACAAcwBSAFYAQgBzAFIARwBCACAw1zDtMNUwoTCkMOsAcwBSAEcAQgAggnJfaWPPj/AAUABlAHIAZgBpAGwAIABzAFIARwBCAHMAUgBHAEIALQBQAHIAbwBmAGkAbABzAFIARwBCACBjz4/wZYdO9gBzAFIARwBCAC0AYgBlAHMAawByAGkAdgBlAGwAcwBlAHMAUgBHAEIALQBwAHIAbwBmAGkAZQBsAHMAUgBHAEIAINUEuFzTDMd8AFAAcgBvAGYAaQBsAG8AIABzAFIARwBCAHMAUgBHAEIAIABQAHIAbwBmAGkAbABlAAD/2wBDAAMCAgMCAgMDAwMEAwMEBQgFBQQEBQoHBwYIDAoMDAsKCwsNDhIQDQ4RDgsLEBYQERMUFRUVDA8XGBYUGBIUFRT/2wBDAQMEBAUEBQkFBQkUDQsNFBQUFBQUFBQUFBQUFBQUFBQUFBQUFBQUFBQUFBQUFBQUFBQUFBQUFBQUFBQUFBQUFBT/wAARCADVAUADAREAAhEBAxEB/8QAHQAAAQQDAQEAAAAAAAAAAAAABgQFBwgAAgMBCf/EAEkQAAEDAwMCBAMFBgMFBwIHAAECAwQABREGEiEHMRNBUWEIInEUFYGRoQkjMkJSsXLB8BYkM2LRFyVTgpLh8UNzNTZEVLKzwv/EABwBAAEFAQEBAAAAAAAAAAAAAAMAAQIEBQYHCP/EADsRAAIBAwMDAgMGBgEDBAMAAAABAgMRIQQSMQVBURNhInGBBjKRobHwFCNCwdHh8QckUhUWNHJigrL/2gAMAwEAAhEDEQA/APllihATDk+VOI8pDmd+KQj3HekI8/vSEe4/GmGubbcUhrmelIR755phjCBTjnm304pCMxxSFcwJNIVzzaTjPNIVzNvNIe57twKQ1zMUhG2KQxg4phGZpxHeK0XFihTdkQlnA/RoqkpBxzWfOaB2O7m5KaErDu4zz3skjkVepxsMssalEk+1W0HWDynJDxp5rdLRVLUu0GKGZEsWI+CgKHHvXHV228Fkf3b4puOoE+VTpampFgZJApdroXvPnyFXXKVV3kSSSBaavavjNXaauiMhXZGy4srPPpQa7srE4oONJ2/7VdWm8ZB71UoPdUSGkWf6f6KBZaUEAnArraCVlYZImG1aVUy0AEeVXyVjW5aaCk4LYzj0pm7D2vgjHV3T1lalOJaTn1xzWLrKrUW0y5Qpq5G8zR7LcpeGxkcdq8h1+uk68l4Or09CO1Dlpi3Jt09IKcJKh5VUpa90KsavbuEr6ZSjgoP2r3o87MPekOjzjOaQ5hGaQjMd6Qj0DikMKYFvkXOShiM2p51XAAqUYuTsiMpKKySBP6Dalg6fN08IOAI3loJOce3rRnQklcipNq9ievgz6J6P6mwVuXVlmVLCyhaXRkoP0o1JR23IqO9tMsZ1A/Z3aRv1vUu3RzAkAZDsX5T+XancYSCenb7rI2Y+ACy2KA4mYXZL4RlLriv4vwqSpQSIbW+WVE66dJ1dKtVfZWipcCQCppSvIjumqdWG14JJ9mRsOKCSPfOkI8wDTiMxzSEZ+NIRnFIRs2hbziW20la1EJCQOST2pC45LadBfggf1za27tqR11hl1O5uK0dv5mrkKKWZEEpTzwg66nfATp/T+nJE63yZLD7aMj95uTn6GnlThYlsaV7lM4loMWU60ohRaWpBI7EgkVgV6m1tA/ce24hSkcD61mOdxxLNZKArt2o1OVyL8AtPV86u34Vr01gUMiI81YDGBNMIINMt5lJNZ2qfwjw5JStZT4QrkKt7hza6KCY6iCBSoq8iLBSU7+8BrXgsEW7DdJHiO5x3qzB7URbuPVjjkNJ4qlqJZCxeCU+m1s8afuwMjGKBpHeqxNls+mjC20tpWjA4xXYUOEIm+FDSlhJKc8elXiQhukZtQPAHsarzYWOQEvzLZbWCAawNbO0WaVFZInuUNBnukAYJrwrXVL6mo15Ov00fgQ2uRA08Fj1qqptqxclHB87DX06eSGYpCPMUhHoGRxSEeYpCNgnOAO54pC4LZfDf0pgP2wz5DaXHsg/MOa1YQVOKaK0fjbuWntemIt7srtuKEEAEIHp7VCUi1COLFR5rtz+FjrnGurQcbsE57bIQBhOCeT9R3/Oqqltd+zByTg7o+rPSnVsTW+lYs2M8h9K2woKSc7gRmivBZi9yuJNfWpEqO4lI2qGSMeVPGVhpRuUk+JjoJJ6lMx2oqg1JadC0Obc+RBH61Ga3qwGUXhorxdPgY17BbU614D7YG4HaoZFC9B+SHx+AZb+EXqQ4lZ+5wAnkEr/i+nFR9GSEnJ5sDV96Aa+08oiTpqYsf1MJ8TP5c0zozQt/lAsvROoGt++yz0bP4t0dQx+lR2T8D74hP0g6K37rBqIW22sqaZbVh+QtBw37fX2qUKbnzwPuu7R5LK3z9m/dItrS9Cu7pkhOVB1sFJP4Ub0Y9mPtmu4PdKfg11FaOpEQ31tl+BGWF7m88nPGQaeNPY7tkbSk7NF7JNza0vBj26LhBSkDCaLuyHeFZEJ/EBrO/OaakQrNEkz3th+WOkqO40OrJ7XtBzdsFCY0F6HJcalNLakIUQ4h1JSpKvPINcnXbvkGhxAG0DAxVElYaLosIQau0VdkHgD5qtzqq24KyHjyJjRQh7SEEelk/vkmsrV8EqZI1uUdvfiuYqoOlc4XmRsaPOKJQjdkWrAlIlYWea2YwwCk7GId8Yik47SCuFdkw00kEd6yqzvcsR4Jn6PxhKnnAwnI5pdPjebF3LlaFtDKYzRIGcV2dJWSHJGYbSG9oPFWWIQXOBvbJz5VXmg0WRrqmEpvfgnNYGtpuUWkaFGdnYhy9yJEWYoFJKc5rxHVaaVKvNVOWzrtLVi4o4CSZCM4qhsUWXp1E1g+dWK+nTyU9NMIw0hGUhHmeaQj0EpIPpzSuJ5Vi5fw7aoH3Gjac5SMjNbUGpwsU4va7k82TUy7dcUvjPhqIzQJRLUZWYn+JLphG6s9OpD0VCftiEeI0r0WORVCo7Fhx3rALfs4OtU233KboO8LUJEBf7gLPOzJBT/5SPyIosJbogYPbLaX26jNGNDTKb/hxk49DToO8EKayfXDZhzWkhQ8QJUCPI098kXxdEoMyYrmm4ZWwglYAPHJoq5yM3gJYWl4P2RpRjIyoelM3klbAK6105ZY7rYfjIHfHAosXdA3yQNrqwaeSX0JYRvV2wkc5qbWCGExJ0dsVh6ZszJbbTaC4suZAAJUfOoW7IUEo5Ce69dHJDqmIzZcHYDyNS2pEt9wm0vJK7K5c5raG3nAV8elBm12CRvywMt0R7VmpzgnYokD2FDuLknvTHS+zx4KUPx0LJHcjPNPfBKyKs/Fr8JMC6Jcvdl8KFcRyVAYSsehx/es/VaVVldcgZQtlEQ6B+GKz+G0u8L+0OnGQ4ogflVOHTl/U7klHuxn+KD4f9JaK0Q5dbS+3GmtpCglCuF+oNWnpIQjddiFSO3KKSOncc4qSwQRpmpEzBzj60hBZpVs8msfWMJTD6Cgoa9a5yo7stRQ0age2o/WrumjdgpcgVKlHxeTW9CGCtIVwHd2BQKkSUUGsIfwnsKwagZcE69HCGm0LHJJq1oFa79xi2eirsfBQj2FdbSwOSJBm7gM1YEbTZaSgnP60GROLAbUCkvrUkDNUasL4LEZWI+vVhEkqG0FRrm9b0unqYtSVy/T1DhwMzOjlt85IFcjL7NRvmT/AH9DQ/jW0fMUV64zjT2mEZTiMpCPKQjKQzJq+HbVv3fdzAdXhJIKcnyq/pqlsFeas7lu4KgtsYOQRkVaku48fAf6Du4Wly1yVZbdG1OfI1TqQuW4Stgr3qrSMjot8TumtUW9Batt3lCNJ29krV5/jx+VAjiRCotkk0fTK+KF30Ew8fmJZ7/hRHgsNXRESreL3aGmSMlDqf70K+UPbDJBVahHtEBs90rAxVm5BhuuQ3DiNKWQEgdzSHIU6vauZdloZZcClJyTg0aKsgM3myIDv81ydKJ3HNTvYEk2xplPvvpDO87B5ZoW8JtvgddLWoOT0JWOM80OVS4SMbEl3y5rdhNwI5KUkAHb5ChOVifOAr6X6d8F7xyOfXFRjlknglefemrJEU64sJIHGaMMV96idQJuqZaozZ/3cHAAHJpt2bIiyvPWvqgnptZipt4fbF5CEg8k1GdRU1dkJO2Cod3veuOs9wKN8qaxu+VtJIZT/wBarpzq8lfLYb6T+Ea53kJ+87q3CUrnw20gkfnRVTS5CKMiP+sXR9/pRc2mTNRPju5AWBtWk+hFDlFLgVnF2kR2kfMKh2E+A20o3hGTWDrHkPCIaIdCGsZrBcbstdgS1FL3rIBGBW1pYWWStMDn3Ct4DNbkVZFQfrO0V7c+VZ1d2DxWA6iIAaTnvXPTeQrJQ6XX1MN0NqVg5q7oZJScSD8lntI6pjpZRlQ7DzrrqfFyVyQbfqhlSAA7+Zow4pfvyHEnC8mojoYn5+9ZJI59aE43JJnNKUPc+dBlBMIpM6mIkIIP6UF0UyamfIBDanFpQgFSlHAA8zVoy7lkekXwc3TXcBqfcn1RWHBlLbffHuasqku7ElKWVwShO/Z8Q1xz9luUlt7HG4g81L0o9iWx+SFeonwea20TveiR/vaMnybG1ePp2NDdJrgi7rkh1WlLw3JWw5bZLLyDhSHGykj86jGlOXCIOpFHVWjLylO4wHcewqToTXYiqsWe2Zc/S95jTFx3WfDVzlJHHnUYqUJXaFJqSwXj6aaja1JpyNIbWFqCRmtS+5EIhzEeMd9DyOFJPlQZIsJhXrbTUXqPpBpxSQqTGWh0KxylaSCFfpVKSs7h2t8Wi0+hQq69NY7bg/eIZCVZ+lSZOLuiOdPtvW6+PR3GFlpK+4Scd8ig2ZJEl3iVG+zRlbgEggmietCK+KSFsk+EIOocK73GzBFoa8R3bwN2M1aja+QUr2wVpuOnL/FuihcYElDiyfmUgkH6EcUWckkAjF3ybO9Pru54a/sK0B5OW1OcBX0oDbfAdIHlaJv8OWov2aY0kZOS0SPzFV3v8ElYKNK6dlqK3vsrpA8w2eKFdk7JBRCtQDm9aTvJ7EdqV7jpEmacU1Z7YXV4ScedHpqyuRYA6x1DIv0pTLSj4WcYH81DlUzZD7e7Be82OXabI/KjxC/IKCRngDj1okU+Qbdsla7F8M2petGunbjqVOIqF4bjgnw20Z8z60JUpTlumBs5ZZbPSnQXRfTqPGtyYbU24rSDtV8qGk+alegq2kuwZRUcB3bOm2jtXW5SI+n4UiKlW4SHG8F5Q/p/5aewzUX2PmB8bvRXU3TjXki4SQ7J03MeUqMoDKYx/wDDPt6H8KBVg1lcFZu0rP6FZWxlYqq+CTDSxLDLIHFYeoW6RZhwPzkrLZxms5QyGbBi6NrdKiAT9BWxQ8IqVHYa49ndcdBKSR51qOL23K6d2FFlgbSBt5+lc/qZu9mW4hhHifIMDPFYcp5CMerBaZCZHioJSPSrekvOakgfGA3j3y4W5IKFq+WuzpXtgErj7aeqM1BCFKOR6mi38k1II2+rT8dG5Su3vUrsfcYnrSl5xKFK+dRwMU913G3EsaQvRnxkLKSdwBGaawRMI3pYbyTwBUbImmfKjQOnvv6+xwuSmK02sKKyMkn0FKCTZRld4R9Guirkdq0RmGb6h0ISBtLiT/arLaDwvbksNBhIkQgUupcXjhQOaQSw5Ri4IIYnQWH0njkZyKnuuQUbEea16H6c1Upb33c008exSkAip06rgQnRjMhnUvQk2RS9sZDjQ7EpxV1VYzKToyjwEejfhk0Zd2m2tTxHG5soBUdLo2x157DcDzVOb3dizGnFcsJZfQbS/TopisadkWcKOEPw5CnGl/Tdx+BqCk1wE9KCzYZNQaFk2WIZ8Vf2+25+ZxCcLa9lp8vr2qbkmR2uJmi7z9jkmOsgsujaQfSqla0Y3Yane9ixWiZUiZbVQGJJSjwxw3wQfL61W0ur0+qco05JuPK8B6lKdJJyVkx1mwJMm1l+GrwbjHJ4xwojuk/WrsuMA0Ct0va79ZS42fAuLGfEYzjeB3/GvM/tp0fVa3R/xWgqOM6eWk2ty/yuUdD0jVUqdVUq6un38EldPLwjUul4knjxmv3bgPkocfrXRfZjqr6v0qjqJ/fS2y/+yw39cP6lLqel/hNVOmuOV8mOdwgsOS3IchtJakIJSrHY+f8AeurMsjDWxNs1FZ7NkltpOdx8wexpuBEix2/GssCS4hKiD4auPwp7iGRx2NarrJjLbQnjxEHbxim5FwJrjcrHJebakW9JewTlHFVZTXFgqjdXBTV8CXHtKJSEeHAcXtTzyKjNtRuhksjHpTTy7jJDhQVAdgBTU1ceTDTUNimNwW2vsiEoXgEqPYeuPOrqBM4THoHTfSrstaEpecH7pOMb1kcfl3NJjpWIYZu8vUTs1+RJUy2+rMl/OCof0j2p1zYi+5ITHUV3TllSLXCMotoAbaQMDHkBRJSigebYI/6yMXTrX04uMC5acSl91tSUtuHntwc44OaruqngeUHKOUfL29dDtc6bvDVvmaYuJfed8FkMsl0OqPYApzVWUJWwivd8Msr0q/Z5dSdXxEyLxJt2jm1JCkNXAqeeVn1Q3/D+Jz7VW/gnUzN2LKvwkJOsXwU9QejkFdxdbj6lsiElTtxtIUoMgd/EQRuSPfkVUq9PnB7ou6FKdvvEJRtPqfWn5d2far1CklhFaV2OatHqQ1vDZB+laFSK2g0jjGthYcwpPP0riuoSaqZLtO1glscIPyUpNc5Xk0sBSQIdlSy0MDiuk0UFTSQF3ZpItpW8Ep8+O1dZTlFRIbWNs6wOsulSDgZ8hRN0bi2sRyrc84NqCSTT3T7icWuRTY9HykSUOryskg9u1Sa8EUix+ifEjw20lOCAO4p7B0FEyQsMnKB+IpNDnzh6L9CtW9Ybh4FhbVHi52uTF5CB9Md6HClKeexTbu7LkuLoT9mpqK1Mty29cT7fLGFYhgJTn3BJzR/SS7k1CXJPWjeivUjQKUNP38X1hHAU60ErI98HBpWCpTj3uSnbUXBLYRNilCuxwOKZoJcRjUVufuDsOMsLcaVtcUoEN58wlZG0kdu9JJyE3YeRpZm6spakKS0p3OxDqQM/Q5wfwNESS7kb3Mj6dFoty7PdYJm2nuh1rKlM/wCYAphGsiAhy3/cVykGfapIxDn7sqz5JUryWPI+dPcS8ANY1fZbnKstyKWpjeWw6R8ryD23DsQai1bIk74ALqNoB7TUldxtzKkxQQp1lP8A9IH+ZPqn+1cL9r56uPTm9Ne1/ityl5/ybXSoUpahKp9B46V64XbbhGnEYbbUEvI5OUdjXgvROqS+z3VYaht7JYn7p9/muUdvrdHHV6Z00srj5lkrkhqK7GnsFJizAEqUk5GTylX+VfWMZRklKLunlPs0eZ2adnyQvrplq1a2eZZUUIkIS+jacYJyCB+Ioc8O4y8BX0buCrdf5NtWshiY2XEJ8gsd8f68q8u6HD/0P7Q6jpSxSrL1ILt7pfLK+SR1Otl/G9Pp6r+qHwv5dv37kl6yyxbo81IIUw4nJH9J4NernKsA+qcdMpFnuqAC6gBKiPP/AFihzxkdBxp8pnaUdSk5Lawrv64NSTuhAv1BYw/AloGCWlJWfXFJ8jAiwhdwuMFbR3rdSEbQexziqlSN5qxYg/hySDr2NaImjRBnyPBcwPCCTyVD2o8opraBTtkFtFdQNN6cbaZfadUoD53ijjPtRFDZEbcmx8nXP7zC7nOUI8YDxcKOPDbHYfU1JYQzyQH1I1g/r2/tsRgr7I0rw2Ghzuye/wBTQXO5KwTK0ZCtVviQXwC4ynxJBz3cPOPwHFM5OOB9qYpiLbbAaiNcA4B7mhbmySSQ4tSn40xlgsOOLc5Kh2SPU1JRkxtyN7u9CjTQ4oMB5pGVPhIw3nuc+tWY4VgbzkS2nVKpboeW8puKFbE5Pzvq9AP8vzoijbDI3HHXvxKaG6WQVwNR3Jpy4ONf/l+CkPyFpP8A4o7IB/5iB9aHOpGm8vIrp8lC9ataZ1Vru53bS9m+4rNKWHGoBIw2oj5iAOEgnJ2jgeVQTTd0is42wIbjZ2m4+AkYxSqPA6iAV6hIY3KSBnNcb1JotRRtYiGnkOehzXKVm1wSSuyVrItm4MDCgCB2rRoa+DxezLcKFxebY0h7cojPoKPX63T06tJl+Gj39hYixNy8ZTnNYlX7VxWEmXY9NFtt0O0XQC1z6kUDTfan1qqg3a4WXTklewZ23RDDaR+7Ga9E0eu9WKdzFraZR4DG02ERxgJFdFCe5Ga47XY1v8RTMZWB5VNkGiOvhe1HadDaLtjDQQ2pTQJIxkkirkWtqsVKa2os/aetdubZGVlXsBmhStzcspseo3WuE+vb9mdPuUYoTkiWThf9dMakbhR2JarU0h0uSFhkqU8APlQCFDCckknOTgD1qKmrjvg6aV6gumUqDPhGGrdtbdbkeIw6PLaogAE/0kg0VPcrshww8BiTg4w2lDq/54j6QhzOPQ/35qNx7CePvjqUGFLfQ2MKjucPtf4fX6f3pciEtxtTFwjubEtpLh5Vt+R057LT5K+nPpSYsEX9R9LSXmG7lHQtM6IBuAVuK0fXzx6+Yqad8Mg01lGaM1WxqOF923DCZCU4StYyCP8ApQZwTTjJXTCRlezQF6y6cytKyHpVqYU5AWSVMoOS0c54Hmk/pXhH2l+x1WnUep6fHdB8xXMf8r5ZO76b1iEoqnqHaXnsySPh+1unX2lblpmaSidCypjfwopB/wAjXW/YzqEp6d9Lrv46Sur/APj4/wD1bt8mjK6zp4xqLU0/uyw7ef8AYG9XH3WdWRHSna42wEqT6KCjn869Jkt0Dmb2Y7afuhhzrLdmsFLMhtTn/wBtR2q/LNcL16hGlW0nVu9Cav8A/STUZfhf9Te6fPdCrpXxOLt81lFi7/CTcLBNa5+ZkkHHGRyK7z2MPsRVqZf2zSdvd7pQoA+tRlmLGCfpvKL0F2MT/wAVnIz6pqMHgmwX6h3xoOQIrbmXklXiAdgDxipT4IpXG/p2pmK/cLlJH7iCguFXvzgUFLN/BPhWQM6n1A/qScX3CrKzhKM52j0qUJXdyLVsHeFotxFwZTMGzYgPKbB59gaJu3Ma1hB1X1oZCBaoq8MtgB0pPC1+n0FDqT24CRi2xn6TafXLusi6lkvIt7ZcQjH8bhHyj+5oMHd38EpKyMni/wB+lu/7u4zlRKgoYOali+QWXwbxVTLLLixX3ymQ8va20hJJUe/f6f2o0ZR7EWn3DO5XZVtg7XXEqlLQNyz2QnFTuIjG6XWRqC5IixWluo3YS0kcuH1P+uKknbLGecD9ffh26raujqj2bU9n0Y2Wtipu1yTMGe6EAAIaHqrKlH/loVScpLbB2G2Mp71r+FPXXQOX94XdSNQWh9W9d7h71jeTz4wV8yVH+okg+tZf8POk9zdyDduQX07eW1oSnxElXsRV2LsiI/3Gf+6754odSeAiQAX6SHFbc8k1xuunuqbQq4uLdPQzIKf6a5jUz2hqSyS3pOxNeEk4wT51Ro0t/wAcuexs08cBL9w4eGSTXKa+pKjWlGTudBp0mh+tlqDOARXPVa242YU00EEeKEgHOMedUPVlGW6Lyh5001wP1pIUlOBkivXfs7rp16MbvKwznNZSUWwgZcKEY2jNeuaSo7ZOWrwzgaL6svNKTjHGK03K5RsVE6fXrTugLNDZu1z8R9CB8q1gAcVJVElZFaMVFZZLGm+ueiJag2i6sIUPLeBQm7h1KL7km2zX9gkxC8zcmVoSM8LFQvkIrNBNbNUWm/QFrtJW++lOdxPGasSjCCywabk8IZW+qtwsTDltkRErUV5fQoZCx5CixqQ2og1K4a6G6kXCZHSi7MtyIKl5aS0f38Uf8qifmA/pP4YoEqkXLARJ2uyU0T1vIbdRKTJA4blJ5P8AhV5/gampXyK3YdY90S6kiSgEq4UodlD0Pt+oqdxfMSXuEYWHCftEJzso8lOfX3/uM01xES6s0UuA8u6WknYn94tpPdA/qHqn19Kle+CFrcBDo7UTWpoH2STtEpCcEH+b/wBqFJWdiSyIo9mb0vq2DeIyCzKjuZCkceIg8KQr14/yrJqaGi9VDWJWqR7runhp+Vn6MtRrT9N0m/hfb+536+6cS9Lh36KN0aW0Fcdgr/4rYvm3kqvyBujJXjRXoSzzg7fYEY/vWVqqMNRSnp6ivGSafyZapTlCSqR5Rae2T2Faeguy32mQ6wkFTqwkKO3nk1bnWp6WnF6magvMmo3/ABsQUXUk/TV/lkh68X23N2SXb0SA4pbhLKkp+QgHIwo8cjkVw2r+3nQdJJ0/Wc2nZ7U37c4uvka1Lo2sqq+23zYm09rliwtoWN5dQlzyGBnj+9Yb/wCpPSYThGMJtPl2StjxfN/y89i9/wC39U07tewOTpCbrJ8RTwS6lW5ZXyOee4qpV/6n9OjKMYUJuL5eE18l3/FBI/ZzUpNuSv8AUdp1xYgaCdhx3d0yVLKpKAOUoA+X8/8ArXUdL+2fR+qpU6VXZOTdozw8W78ZvjPkztR0rVad3lG8V3WRs0C0zL1VHVLThhkKdwocLKRwPz/tXT6TqOi1ctmmrRm1yoyTf4J3M+pp61LNSDS90OWrNWCIZLyHAqQ+ooGOceZP9h+FaSmllAHG5DdwkKmzSCSrnnP61RlUvItRhZFgentgcs2h2nW8JccV474H8WMfL+lHyoYBcyuxfIvELUTHg25sm5oOCUpwk/4jQ1UVZWjyO47HfsDlwdZhOOTJCW3zGBQ24gYCnCMHB9B2qxTi4ZllgpS3YRGN+u8i6yfBaKnFOq7DkqVRk7gmSz0+6eRNMw2pdwwqcf3rgJACcc4J9BTSld2HirZHS9dbLZbXnI0TEtfZTgVhsH6+f4VNU2+ROXgHVdSLrrpblmi29V28ZO12G0wFI2H+vPAH+I0dbYoHl8FXepHWq2QNcy9JaD6TaZ19JgENS5cC3KmR2Xhnc1lhCQVJxgnfgfhQpSd7RiCbV7JDN1d6X3XUujNO6k0j0v1BYL7IcfavVhiMreZjhKUFDqEElQCiVYx3AORkZNKvTqShdRyEwrW4KySy+3NdYlMux5LLnhusPIKHG1DulSSAUn2IBrh68ZxqP1FZhL3V0GOmZCWkoGfrWG4bqyuFg7ImXSGHI6XCfl7VsQ0alG5oUqlg5jxG38KSTu964vq/R5VZOpT+8b2n1CjyOC4yoyNxTx6iuBr9P1VDM4Y9jdp6mL7njdwQgYzWa6UpPAd100LYd5TCSNxGSc16z9mtHUoUvjVr5Oc1tVSeB0Z1O2cZUMV6rppdjm6r7nGXc25CspUCD6Vq3KLPkjPuMm6P+LJdU6s+aqSVij8xMR7Uhwm0dbdYahmiJpaHfLrLPAj2hl59Z9trYJ/SpqMnwhrMsXofp98Vek4rblu6cawVGVghuRZ3Ek/UHB/SpenJhVvjwyTomseulh2u6q6G6ukJx88mJZn3kgDzO1KsVF0pIIpyXKuJHvi8sFtlpbmW+fYZaThbMuKtkg+4IFBlCS4J+ou6Jm6YfFJYb8pH2a6Mrc4yhSsFQ9CPOoKpKHJNbZ8MszpbU8HVEESoLgdSOHWc5Ug/9D5H/OrsJqaugbVuQlgSQx/u7o8WI6MAH+1TGENysioDwcjklhXzNLT3SfMf67inyIBNQ6PdRIN1safAuDJLjsRvgLHcqb9vak8objgXWe9R9YW8KX+7mIwHm+x+o9xQW74ZP3CSIiLetHXTT14fZivREl6O/IUEICfI7jwBk/kaHKrCnG9WSil3bsvzJKEpO0VcgiNPi2O8KZeUqO4lS2VyFjKG1pO0oIHOcg4Hn5V8/wD2j+1/V56irp9F/KVNuLtZzdm1e7WMeOH3O50HSNOqcalX420n7ceDxrVUuUww+ZqnW3krKPGdUBE2jBSokfINx7nAzkehPl+q9bUztqZSk42WXuum74u3f5L5nTQpwgmoRSQhXqQOyGA1OEfxY6XVIcWFBKwobdqsnnYMk8Y96GtLti90L2dseLd8LvxyWkvYWRtjakMPyAtbym0trbWoFtKmyRwTnnBTgnjI9KBO7vOKwr397Pyvxv3LVNRZtdJkiDbFG3xn3XvlUClslSAOCoeqcEDPkaajTjUqfzZJL5/l8/7B1sv8Q72d19/c68ghT4G9SiMIV2PAPr6d6pV4wj8MXx+f7/IE0rWHNUhcZe1vdggbsDBwewH40CjOdOca1OTjJO6aeU13vyBnSjUjaSwM8y2JufypV4EpIKQFn5c7jn6c+ZNendO+3/VdG0tW/Wp+/wB7jHxd/qmc7quhaepd0vhf5fgDtisj7uoWIjrStylkgkcKSOSR7cV7n0nq2l6xSVXST3cXXeLtezX5X4fZs43VaeppXtqK3j3CzVatYQrnHjomt2mA/wD8d0KO9LY8gkdz5CtuVKpuu5WM5TvhIdIeqPDjot1uBgWtoDxnicuvH1Uff0q7TSitsVZEZ+4zajvhngJSC3Ea4bb9fc0VzA7Rw0DY2LcwvUN0whtGSwlQ7++P7VK41vI0aw1xcNWSlNMb41sbV/w0H+P/ABHz+lEj7EZAE/qmAi6OQGy5cbg0NzkKCRubT6uuH5GU+XJKldkoVzh3VW7ZDLIe7FNy1xdrhYnbKQzAszmfEt0Hchl4Hv4yshT2fPeQD224qzCK5kDk2wdl680705s8VzU2oYmnLI0AiNEQCgLA42sRmklSgO3yoCR5qFWHJQWcEPcU6d/aAdJrI42wxa9U3FtJH+8ohstp+oS49u/PFZ9XW0abs2TQm+KXqH0t+InpENb6YlKVqywzI0R8TIZjTTHeKkht0nKXUApylSVr24I+XOKx+ozoarSynB3cf79h2mmmU9RfPu51PPGea4uNFye5die62CStL9T2G1NM9uwANblKvC21qxNTaJp07qRqShCgoYqU6cZl2nWaCtd2bcjjBBzVWWhjNcF1aiwO3G8tsKIBGaDDo1Ldfah5at25B24aoSlOQsEjyrSWkUEowwU3XvljErXhS9grwAe1XqS9JFWdTezZvqQBJQjxOMgd6lLUWdgfJ8//AHrVKIQ9PfuM60tJ1InfZEvhUlJUUpUkAnaojnBIAOOcE1ODSd2Sjl2LFo/aI6409A+6dGad09pGwtgJZgw2XMhOP51IWgKPqQkUeVdvCWCV0uBPF/aD64W6VXSw2S55OT+8ltn35Dx/tQ/UY+/2JN0R+0Q06t9KL9pa72BWR/vlpnJmJ/FtQaUkD1ClH2qSqIkpplidLdftEdcIgt0XVVq1WHR/+FXxKVPkenhSE5P/AJc1O9yYE62+CTpvqhxcm1R7h08vJJ2ybMd7AV5FUZw4I9kKb+tRcYyGcUyOX7Z15+EmaL7ERH6haUiqw5NtwWvLXmmQwR4rY8icEAjIWeDVf0nB7oivNK3JdroT1v0r8Q+h2tQaakFIBS3NtzywqRb3/NC8dx3wscKAzxyBZ+Qk75JRitKVuhPDAWMp9D6Ef64pl4HGm6WdxKvGjkpebPB7EKHcH0P/AM0z8jgfedJquWL/AGIoj3BCtkqKOEuH39CfX1ocvjV0OsEM9X+qUq4WxliClpP2IKZltvAJdcWFduTkpSDjgZznvxXjH2p6zT6jUj02m3tX3nb+pN4v7fT5nZ9J0To/9xLnt8rABMvTcRmG7JktSIL6g482FBLclGUkJKclTaiQEpV/DtA8ySPM4UJVJSUU1NcPLaee+FJWy1zf8DrI4SQrmCKq53iLEkTRcCthDrS4iQplKkqDiCoOBLrY+XBODnByQDkUdyp0pzS22lbLy1az4bi39VzhYJxbaEBiKubURlaB4zzkmKyI8ZxDjzzZUlKxuJCVAghSc8EAHI+YH3qk5STwlFu7TSUrXWMteHbPOHgOn3HSzTHXWoioe1wKhplxAE5cmBCjkE+RIAOPJSkp7GqdeEU5Kp/5OL8Ruv0XF+6TfJYVrBHHnCUy24w+laCgTFOAkbEqU4EN88qJUlJ48h6kVlyp7G1Jf/j+FrvwsX+oW/kIoFwCLWtBQFPsNI3FKjgKUoYyMDtxkH196zKlNure+G3+Fv7k+JDi1PUmU60QQhRQkEZG9XBJA9Mc4z71UdK8FL5/RCsmjULalBIWAQvcrH8oweR7kY5p7OHH7/fYi8C60TnLZc4Vx8JMh1pXLaycHPdOR5Hit7onW9R0LVrUad4TV43xJLs/zz25MnW6KlrabhPD7Puhq1Fdpl6u7r0pe95w8AH5Qnyx7V9RdM6vR6vpIayk/vLK8S7r6f7POK+llpqjpNcfmvJxLaWI5K1YQOTg9v8A3NbsJ2RRkrs1s9uN8lqdfJZtzHzOue3klPqo/p3oyeSDQr1PffvRola27faIqe61hCEJA7knj8TRo/ECeMkQXy7aj6iSU2jTLUjTOkiMSdSOo2SpqfNEJtQylKv/AB1j1KQeMycty2wwvINxb5CK1WCz6HsiYNujIjRm8kNNkqUtXmtajypR81E5NHpUlFbYqyIt2ENsuVsMkLkQ5MxSTkJddS2j32oAJ/Ekmru1LuBbbfBK9kb0ZrYpY1DoW0TGRtaDk22oc+UH+tSMgD6moOnEle/KArr5+z40VrTSki99MYCNMaqjIU83bo7hVCuKQOWwk/wL9FJxzwU85GfqNLGrFpLInG2Ynz+gR5ENDsZ7xGVIWQ4wokYWkkfMn1ByOe3NcFXlKLcGFWUhsvZO7Hei0AUxTp9am321qPYjmnqNKSBxbJw01qVuPGRlzGAPOtWhaVgqnYJh1BYaQEh0Z9c1tRpxsE9Rgze9cJeWoocz6c0RqKRHc2Bd11g6yhaw53HrWVWmoZE3cA5OuZrkzg/L7VSlOcle9ge6wrZ1MojxCtQV35rLqRqSldssxkiCa7gomeXvSEe5pCJl+G34Y798RmpFxYk2LYbJGyZd3ncpTjGUNoyC4v5k8AgDIyRkAmhT3ZfBJK59GtCfszuhthtjLd6ZvGrJagC5Nm3JyKhR/wCVtjZgfVSj71Y9KPgJtXA63z9mJ0BuySIkO9WR0jKHIN+Uop9CA8lfI+tR2R8ErLsGujfhOuHTi2tQtO9UNRXK2M8Jt2rWGLkyE4/hQ8gNPNAeQCikf00nHGCSuFLmhb9ZwHUtocWnuqC4Vfkk4Vjg+RqGUh7XI0ndHbQdXo1po55rQnUFALblzhM5hXRv+aPcIgIS8hXmtOx1J2qCipCcJNPKGaJg0lrI3wps99hixaibQXUxvF8Rt1IICnozmB4zWSArgKQSAtKSU5TYgf1J1109CLivGUqew4ph8IT8qlJOPmB5B8s0Ob8ciIa1P8Q0lsXM2ofY3JrBZC46slJPG9ORwr/5HNcj1jqi0mnk4S2yldL2fn6efc09JQ9eola6WfoV71ze7zqOZJnvuKcfcGHnkpSkuqxtyQkAEnAye5PJ5Oa8qVdams62oac5W7W/Tv393zm529OmqUVCGEh2ul3F3Q2qLFuc2MLdm3rYbYbUksEbipsDcpCcZO4DgLwoYJGPQoqg2qrinu+K7k/vcZ4u+MeVjsXd9+B6kwrxdA1KdaavcicpFztk92GylUhTSEFyOtSXRsOMfKrJUCDg44owlQpRcY3hGPwTSlKy3N2krxz34tZrlFhPNjazYuaYkiAVNxXZwegtR3HluW91R2ux1NbitbZKXAFclJCu2QKhX/kuUKuZKNpNqKU0sqSlaykk1jhq3OWWIS73FtskxHHLO+phCDKdlECG+pYW5gJWWm0kFC1fO4CjAwVYAIG0FaFRKpBO+1R5SWO12+UsRe67vbLTd7Ckuw8wZ0mPGZivSXokqOyDJDqlpDZTuyyoEAIWojcM898dzVCpThKTnGKabxa2b2+L3SWPwCKXcf404llHjoU1HSXJIUUBQajJUs4UvncUDuk5yUZ57nNnTTk9uXhfOTS4XZS7PGH2CKVh1bnlMcOPIDYBVIOWztbCuEpzngHsfPgVTdO7tF37fO3L/v8AiS3HVp0tpZUrKkJYytaAMDco4/uO3nQ3G9/n+hFyuLUOrcRtbbU43sRyjBJ47kd/p60BxSd5PyDdu5znKbjMokrbCFISAtRONqTjHH14zXpn2G6rHS6mpo60/hnmPjcuc9m1+L97HLdZ07qQVaK45+Q3ILt9Upe77PCR3cVwPr/7V7tT1CqfdZxzp2HNT6pMZuFESpENvsnzUfNR+tacJXVyrJZGu7CIdjcpLTyWVApS6NzaFjsdvZSh5Zzjv35q0m547A7JHEXCL9iel+IkNg4XLkq+UH0A7qPtyT6VcguLAJPyDq7kq7OK+wRCtvsqXL+UY9ucJ/U/SrqpqStLN/H+iu5NO6C3TlpfTgoutvirXyS2o5J9yBz+dHysJEOSUNM6eviAy/HvkR5IIwFKXg44waDJt8hErElQZb0OREiyC006sKcQpCtyVnAGM9/P68edBbzyEwfMP4x7Hb7H8QeqmbXbkwI58B97wiVNqedZQ6tX/KSpZyD51wvVoqOqdlyk/qRXgrvchvXg1WpOyA1LmRXwyR2GKU47gabRpctYPwGlBpeDj1q7pYTjhMTYwJ17PH8bpIPvW5eY9zuxrV58/M4e/rTyqSS4HjI6yL4uUNpVnNZEoyk7yCbjiyPnBPfzqEuCI5NkFs5H51VfIeJE4rsCqZSEZTiJG6b9ddR9J9M3q26XkLtdwubqFqujThDrCUgAhsYwkq5BV37YxgGiRm4rBNNJDjD+Kfq1CcCxry7yCP8A94tMjP13pOab1JDb2Hukvj46oafebTNNo1A0TgtS4KWlL9tzRQakqjZJTb7FnOnX7QvTk1tkau0hqLS61fKqbb2hPip/5iPkcSn6BZ+tFU0widy0XTbrfYepUVD+jNa27UDeATHZkAvN5HZbK8OJP1SPrU+Rw4nzWbyjZdYe2RgD7ZGAS8kj1B4V+J/KoOKZK5H3VGBBZ0hIRqFp6fp9C0ut3S2lSHoL2CEOpUPnjujOArlJBKSSkkEMns54JYeSpupZ7yVOOG8LvD+SPvN1CWnZQzwpxAJTvIxnBwTnFZWo1Lpq0sryufw/sSjTT4eSPbff3JUmZ+6+0JyGylK8EHPPB7e/avNetVXqasbysl7P93Ok6XD090rHN2W1AnxJ8a6TbeW3CouI/gCwP4RwRjgDNY6hKpTlSnTUrr8jbdlJS3NfoJ5NzbIvG1XgLQ2n7OxLGXWxySUKzwOw4z5ZwKJClL+XfN3lrh/Nf5DeraTz2x/oW6YlMR9HLWVvMamdcR92MuxQ8ZSgofu0PZHh53KJAPYjcPQGrhKWrSsnRSe9p22q3LjbPa118n5NSq7aScvvPi4bp1pGZZi3e/GJEfu8vcbowksyrc60gIJUjaE5UU5PfKiQTu5rB/gJycqGlu1Tj915jNN3w73xf2suMYLbqxgk3hP8hVB1CzdmkLjlyGzdn1RX4kRsrfEpr5w+hByQlaULUeARjuNxNBqaWVFtT+J01dN4W2WNra7ptJc/WwaNVNprCf43HO038alERq3rjMzLmwmW3bkTS2kvtk5ddCknbkqaAAyMhWdpANVK2m/hd0qybjB7XLbf4X2jZ54ld82ta/AWNTclbvlII7XqVi8KjSHQuXAlvgruIi43yCdoZBHCwdudh80kAnucutpZ0VKEcSivu3/p53e3PK7O9lwFhUi7NPD/AHYdHLkp+4JgRA0p+WlLqleIW0R0ZUks85V83PGAMZJxjFVFSUafqzvaOOLtvD3eMeb+2Qm/4tqz3HZF2Zfiu4Q2Xd2SwlzJbQng4wcEZzxxkk4zVJ0ZRkvHny3/AK7+FklvTQ+WxG8ocQlgIWsAjcQUq/vz7duKz6rth3x+/l/kd4wbuqjohvtyW1KY2qykK4UEHdgkdjx3881Z0NX0NXSrNX2yTtw/oUNVF1KMo3tg9tVtuF9YRKfaTb4uNzbK+No9duf1Vg+1fS/T/VrRU3HauyfP7+f4HntVxi7cs53q/wAKyQcIWdpBxg4W8fY+Sffz8vWuqpJJJv8Af+ijL2I8kT37y4txW1lhPG5R2pSPQDy+nc1owTk7RK0rI3i2R+5qaCnfBjtg7HZXCQD32N9zn1496uqcKaty/YrNOWXhBrYtP2xDjZkBcsp4D0x8NoH+FGQf0p/VrSwlZDbIL3CSTCnIQ2iywbe23jlSFx0E9/61f6zUJRqy7kk4rsFGmtIardkMvz3PsUJCQsqjy0rKs/4f/imUKi5Y94gxr/qLEsa5sVVxU4koLSpLiiAgHOSFZ4IHp5/pFz2/eZF2sV56n/E1pKNp+dZLJa0X+XKQpD01/lGSOSVHJUfz+tY+r6jRUXTgtzf4f7/eQbd+Cm8p5KVknyrAhF2sBl7DK9dEoUrn9avRpNggfuU7x3SAcgVpUqe1XJdxAtJIA71ZTGszrDbO/wBMVGbwSSY8sHaRVCWSfsObKgpNVJInbAtaUAkc8GgtZDRRFddaVGeU457TDGAcc96cctl8KPwOP9ZIMTVuuLo5pfQzqiY6WsCbcgnglrcCG288eIQrODtSe9WadFvMiSXkt3a+rPwrfDYHLZpeLp03aMfCclBIlPBQ773iFuLOfRQGfIUT4FgJwLB+0P6ZzlFDmpkx2U922YDiGgPf92T+ZpboMW4eLP1N6A9UJzElu5aMnXZRGxclcJuTn2LiEuZ+hBpWTHuTnboERMVKIT5CEj5E+IpSSPLBUVf3NM0OazoTimXULjtymnEFDjKkjC0nuFJOQag0xylPxCdNbXCuL7mk7g/apoyp2zSmSfDB/oCiCtv05Ch/UsViavQ0q3xK6fs7flx+QVVJRKwIupslwfbujxZcVjetgqTjGRtwrB/HFcVqNHJPZTXHlfmbGl1EIX3u1/ARQ5dxbQBBvNvltqSoCNcUFKmyU5wFJUFcnOMp5/GsWcKLf82lJPGY9/o1b8zajKorenUT9n/rP5Cl+4B6BJfZjIvSWFsvPeOsb21JGSjkcg7Tg8/U8ihxpOM4xlL073SssO/f8/8AgI6t4SaW61r8f4O1zutyhWDwJd6Z0vAiy2pcW0PtB10LXjetshQSggckYPPHy+cKNGjUr7oUnVlJOLmnZWXCeG3fs8fXs9SrOELykqaw7Ozf0yvwz9Btm3SzPW7WFvdkQLtNHgXC3XlbmJSFoIygbMIUnjOABkn3GLVOlqIz01VRlCOYyhb4bPvnKf6L5ZjKdOXqQc1Jq0k+/wCWH9AjtN0VKjXG7Q4LdxvNhmR7hMvsCUlstQVApdYQ2VELKmyvjvwFA5wKy61HZKGnqT206sZRjCSbvNZUm0la0rfmmrZLzrXk6kfilFptrw+Va7za/wCp31VrHTi+nhAan2habwq42ZgxVMPXOEsjed38WxRW5kpVjDSAocggej0WsWv5jP4FGbvdQmuMcXSStdf1StxkVbV0vRUsqzullOUf1s7u/wAkSTA+1SY3/ddjVEVdmkvaejOykiLCW222kuOhspGCVbigbyQspBByRy1TZCX8+rf03aq7PdJNt2je/CVr4SavZqyNdSnttCNm8xXZfO36ZHNF0ZSXGXrqy+Jbqvv+e/HShthxKkBDaVoIRlRRjlWUhGMneaqujLEo02tq/lxTu2mnd2avi98KzbvZbSXqR+65Kz+8+M+PH5hXAvLqENPRJFoXa1qXHYXDC1SHkf1BAG08bux+UY/iJzWPUoRbcakZ71Zu9kk/F73/AMvwsB/UbtJW2vGP3b99x8ZvMZxxJbuhUl0IBCGgSE+2D6geXNZ0tPO1nTyr9/32JerB/dkjjfNYN6VtK5a9sx9biWmmWwofaCo5IyOSUjKvbn2o2k0j1GoikrKOXftb/PBV1NVQpP3wIXuo0udFBdQ1Hax8lva3bU+7hJyfp/avdela6ervKVlFYUV+rff5HF1aahjuMKUzL7KL7ylrBGd5HJHsPIe/avQdNTnVd3x5MmpOMFYdEww1tBH8AyEgZ2/h/nW9FRpqxntuTuOkWxzJOHHdsRo8+LJJzj2T3P6VF1orER1DyP501EhQQ5sXKkODhb/CR7hA4/PNZ9fUzStEs06UeWJfsyY6Ut4BUrsmqTm7ZYbar4Dn/bZzQ2n0piPpZCEFTqnOUAY54rRpzdOKRWmtzPnj1n6yu9Q9SPQbW6W7LHWUBaTy+rzUfb+9YmurOXw9im53fsAik7WvoK5+92RuCN7mFnfzzW1p4bivJ2BCVcSSRnFbUKRBXZpGy4cknmpSwEQq8POO9BuTFLDJQN2aHKVydhWkkD/OgMbg6NSFNk4qDjcdMXCSohI9KDsQaLsiOq6Ur8mYpCPdp9D9aQ1zo2yXDjHFPFXYzlYsbr/4iFah6W2ay2KTcLddHUGJcWVZS3GjNpSG22Fg42ryrgAFISR/NmrcqkpY7C3RUbrk1+Hj4K9UdeG27vIlM6W0eFlKrrMRuW/g4UGG8jcB23EhOeBkgioKi3keGVdl1LT0e+GT4Z9PKnXO0WjUElpaYzl41htmIU/gkIQysFvPBOEN8Acnii+nGCuwya4QnPxedLUxlxoOodIWWGQU/ZbZYg2jb6EJi4NDUorgfdflhDpr4nem0wIZVr+xlIGEoVchGwfLCXAgCpbkxJrklfTPUXT+o2R9z6gh3Hb2EO4syQPwQtWPypx7oXap0zZ9dWowb9amLpEUCUqU3hbKj/MhQ5QfdJGfPNAqUoVY7Zq5NNrgo98Q3wmah0cl7UOl35N7scZXiqaQXHpERAySVtYUVIA7qQMDzCRzXNarp0oNuC3ReLdy3Tq5TXKK7Qpsa+sNi42iBcStJysEIeIHykJynueDwRXO1Kc9O36NSUf085s/8mvCarperBP9f0CG73O4TGxNjttRihaWX4CQW33UJI+VSUkj5h2xkc+XNZlCjRpv05tu+VLlJvum/D/bLVWVSdpwVrcru14+osiTo5uyTDkuoiT2PsMiRe4avCgK252pUsAIPITlJKeU5IwKDOnP0n6kVug9yUJK8890ufOVfmyYZTSqKcLpPD3LEfx/tjyazL/qKfcbXqY/c97ct8gWhgowFSFhKUIIUTtScFIGD8uD2FPT02kp06mj+OCmt79k228Wu+978/MBKtXbhqFtlte1e/bD/wAGj1tej/dsy+2BFtscKSLbd49oeW27NC8bElLYRvAUtB25JUc+pqUasZb6elrbqklug5pNRtzl7rYTV7WWCc1JbZV6aUIvbJRfN+MK11drHLCe2an1VYoTF4dvlvgwbU87p9q0XyMkSYkV5SSVOoABUohDRVuPzJT5gKrJq6TQ6ib08aUpSqJVHOD+GUo3wn2V3K1uG/NixGrqKaVWU0knttJZSdrN92/7D7pC0WO83O66csU+96rvcOQwzb7jEu/2VgwyEFxrJWlsJ+dxGEYyoJIUBnGfrq+qoUqes1UIUaclJyi4bnvzZ8OV8KWeFdWbLFCNKpKVKnKU5RtlSstvjm3tj8Qqt13gWFgJnwFt6JvgdEO1RUh5w7A0pSHCMhLZVtI52jOCrgA5FWhV1Er0p31FK26Tws7ldcXdrri/dLxeUo00ozX8uX3Us+Of3b3CV3VzOnU264XS1twJ86EluIm2pcCcJcUBvUlASVZJHP8ALt5AwayloparfSoVN0ISbe63dLhNtpf3v7lt140bSmrSaxb/AIFV110zZYsm5SFwLc6tsuOxWFBS3853OOLSAFKPOSefcnFAo9PlXlGjDdJJ2TfC4sknwvH6DSqOCc2lG/K8/OwJ2TVN+6iOOMWgLasrSwoqIU22FYIGVLwBgE8E55zjtXZaX7N1Zv8Al091Tu8WXzfBg6jqEGtqdor8Q8tGmoduAM2X9oWO6Y43D/1HA/LNerdH6DDQ016zTl+RzWo1nqP4OA+ter7TbGfBjW99sEYUtrClq+pIrr1GK7mZuYshWXTt1fMn72uEJ3OQ1KQlKEk+YSEgH6kk0N0oy/qHUrCiZ07vbjjcu2eBqaIk7iGJSGVjHkUrP9jQ/wCFm/6gnqK/Ati2/UV0WpiTpidaShJIXLW2GzjyCgT38j296rT0dTmLTCxrx4asMotkli4rRNYcjvpP/DcTg+x9x7jis3bONTbNWsWm043iVl+M/rQbPDb0ZaJOyXJTmY42r5m2+xTkdie351d7XMyvO3wL6lWNPjYlPFYOpyU7j9KfAYVkgcVnQj8Q265HmopOFK5rptLAg0wXWdy81rLCJLgd7agKbqnVeSURyaZ3EVVcrBl4FLjWxvnzoSd2JpWOOeKkQNm0lRFJuw6QtabOcmgNh0AIBUa6IqvAsjxSupxjfkDJsd4dm8TA25GaKoohkfomkQpAUEURRtwKwvRpZQChgjI49qi0uSSTLm9J9bLvWiLTb2tQw7LGgx0xG4Dr6WlJUlOFEAkZPGQfQ0Z11HhFiELrLK69Z9RS+o2oGmkurNntm5mC0TwewW6R6qx39MCqVSspu7H2vhAxYunk68zGIcKG9MlvrCGo7CCtbivQAcmhqSbshvTZa7pn+zM1hqmOxL1NeIWlI7nP2VLKpksD3SlSW0H6qJ9RRlTk+cE1BLksNoD9nhoDpRcDd3tSXudMW14WS1HYV3ydgSgqyfrT+ku7CRSjwiS4HQiIwtDls1XqazMD/wCm5cWnQoeXyqaIH60nS8SYRNd0Ol36MxLjB2HWN8akAcSGJDKVg+uA0BSlTclbcxrrwU966fANeJ803PS+poNxnqX4ngXiN9kU4rIOfHZCkk9/40Jz/VWRLpuX8bd/Zf2sv0DqpFNNq1hNZ/2d98lwGrhL11a4GoeXEqatTslpvKcFIWp1KiO/O0d+1Vl0Rbdkp/D3Xn6/6Zbetbe5R+Lz+/8AJC3WboVr3pa2P9qrkTZ1rAVLtzIXCkLKjgFQSVNFQx8qsZyeeOcGp06Wim3HT38Svey/K9vdfQveutRGzqteVbn9+xHH2iPLu8oxtIpzcI//AHc03IQFxlju6sADbxgjHzcAcZ4obZwpR36n7j+J2dmvCznx475JtJ1Hto/eXw54a7+36jtYNP2e/QHFIcuF61DIZUFJdfLamJoJC1q3kDA4G5IUcJwMk1S1Op1Gmmk1GFJPsr3g+ErefDtl3dkg9DTUqsL5lN+XxLznx5z4Q6G8SbBd4Fxul6tt4n3mG5Z7qblGKxb8Jw2pQGAnlS0pz35JBFVPRhqaU6NClKEaclOG123+V3b4TduOE7hnKdGpGdSabmtsrrEfH9+RXDdRJtibPCmwo12szTj9tnsNONp1CgIytgKyNuRlHG4FeFD5DQZxcaj1FSMnCo0pRbT9F3xK3e3ObNRuvvIKswVKD+KH3Xa29Wyvrw8vOeAzVrOVo8RtSmJb9L2qVi0S7HFZWZEZ3DhU4AMY/pUAAQSggKBKhhLQw1u7R7pVZx+OM21aSxZd/mub5Tawnadd0Eq0koxlhxSynn/h48PJ5M1/dY9lNg01aZ+p4cyO4G3VR3FmP82GsuK/jKQePlBBT3OcizpeirWaj+IqtQnFq6TWfOFxf5tNPjAOtro0aeyn8Saf08Xf7eDtozotqbUb8d/VDkGOhnGGZEle4+fzNt53H/EpIFd3S6FXk2qFoJ93l/S2V+8mFU125JVXe3ZE1RdFsQURo339DjJ4ShKYZCEeoACziuu0vTqtGCh6i+kbf3ZlTrxk72/MI09K96ylrUkd2QAFbHI5HB5BA39q0VQlHlgHJPsIV9PpDUkNvahtzYPGAshf/pVgfrVhU1FXlj54IN3eMjgemz0ZO56ddXEDBKoraHEY/wDIomrCjFJNcA3fgJNIWC4291D2mrsi5Ka/4kJxSmXj9AokE+2RRsJEck6WyeLhaWJbqPD+UGQy5wtryJx54PB9Ac0B4CcjT1H0sZelbhMht+LOhx3ZEcDGVFKCopH1xj0zg+uQVaSqxs+ScJODwfDLUWqZuutVXC+zllUic6XsZzsSeUpH0BArIrYwVL3k78j/AGdPhpHljzrnq7uyVje6yFJaPlTUYpsg42yR/eXyte30ro6EUkQ9hqFXCQ92r+AfWqFbklCw9RUZUO1UJsOvJ2ltEo7VCDyO+BIGjkelG3A7HVkfPioMksDzEZBSCR3qjOQeJGLXcV1pQkENnhF8jgY96sR4KuW8BzZ7OCEgYyfSiBLWwSJYdKuPNJJb71O6Q6TY+q0QsgZawD5iqVSTLEYHn+wSgoFVZlaptRahC4rZ0K2k57/UVkOrNvktqmkiw/wwTNNdN3Z95ucUP3dw/Z4pxktp88emc8n0ro+nJSpObeblStiSSLHay+Jex6SYDQW7NnFoOBprKEJ9B6n8TV51qcW03wRUW0mQhJ6s9WtfTVvWOPHhR3M7HZLZUMeuVHKh9E4oPrTm/hiK1u4WaE/2yv8AI8K7dQShDDXiSnIMFhLbaR3+daD+iaLGnVk/ilb2sLdFLgl5Gv4FgsbTsd4LaDRWmVcFYJQkcurxgJT/AKxVvZblkNyKuah/aEaKvGpXIDT0m8tMEtLlswSmOvnko+cEp9Dg5rO1Grp0Fd8CUle1x80Z8Sln6gPrh6R0zq26zm14xabcpxtB9FL37U8c5URxQYa6FVXhFv6E1ngLepJ19crRBtLNphRlXXLElOoVtuJQwSErHgJVlxRB4BITwSTQatWtUlGEYYfN+30DxUY5bA7rh8Kz/VTp7pG69Mo9rtMawMyIrTAiLjz5aQvwdjbpKR4ZKCvLmSThQPJJr6/Tb6DVOCft8n+YWlVlCe6Us/5RVmf0C6sWqTcDfdGTro3Y4BH2aIksLDQUCHkKQf3pTgnKMkAFQGc1yb0jUpUqMXCTe5trcu91ns/+cGjDUysnV+JJWSTt9RP0/wBA9RdRaPmTYVtk6iL7raS0pbJy2kHASk7VK9FYB3YFVf4Klra3/bQsoX4ve7+v4eLlmnrZ0qT9V3k/NrWQddAvhuufVSFFgXPVUmxogrK4zcaOlxyOR2CApQ2DsnndgfLwOKsaLT09frKn8u0WrNv+p+67+W8ZzyV5V6lKkot8ce3y/T5ErdNeh1it2kNTac6lT0ydQt3Vb0OfJbW4poJACF71A5DgKtyc9iPQVKn0vSydSNR+jNWjjCsvFlZp/j35ITr1ZRj/AFLn/n3HyL0clgrGnHYT0ZHZLExJOPccfqK6XQ9H09FbqS573v8AmUqmpk8M7QdPXTRkxD15dlwEKGwPJabW0SfLcQpP5100KMI4KDm3lj7qfQK7/pt242h77fIYHiKYS2hDu3zKdgAV9MZ96tRiouwKXxo46Thsak0kDNZEyTb074z3I3t/zIUB6d/wrivtpDWrpNWvoK0qc6fxNxdrx7r8M/Q2+iSovVRp14KSljObPseRoPhSktojNpG7O9KAAfYHzr5L1Gtr6lOderKT95N/qz12nRo04/y4pfJII2mwhaS04WVp/mbUUkH8KL0v7RdU6NPfo68orxzF/OLwUNXodNq1arBP37/iJJ+qn2ng4ooF1ZUPDmpSEucHkLxwsfWvp7pP2ofUtDDVSgk5Lt5WGebajpyo1XTTHpjqLeJd3jzVJDnhpIVGbGAsHG788dq3aPUpVaik1aP7yVZ6dRi13OfUb4qNIdKtDaldnXVkvMRXEW6CVhUh9xbWW2koPJIKtp44ABNbqqR5M94wfMvpb8LOtNZwo816Km0QVJSUuS8hahjvt/61i1U5/dIxpyeXgnS1fB4zGZT9svj6l9iGWQB/nVL+A9R5kE2JCq5/B3p6Qzj/AGkuEZWO7jSSP/41cp9PhHKkDcbkS63+CnUUTxJGnbpFv7aezP8Aw3cfqKvLTuKtF3K0qcuxXvV2jLxom6fYL3bZFsl43BuQjbuHqk9iPcU1pRxIjlYZztf/AAx9ap1uSUAktrO9Q+lZVWVizEcJUYbKrwmSYiMfBHHFG3ETj4Xhujng0S90LuO0Vz5B7VTmg6ZGDY+YV1rM6QS2CWlCgD/ajKWCusMmHQkFuW+2tXIyO9HRPlkzW6O202kJwAPSoy4DxQ/ww2sYOKpTZYihHclNoJxjGaw9VIu01fg0gqQ4Qc1mRlcsMci6lsfIognyHmavU5yWIPkFJJ5ZNmhtK2DTaIs6+oTedSvJDgZkKyxESf4N39SuR7Dk9q6Wjp9tt+ZFOU78YQz9XOuotFtlswFoagtJKF+EAj7Ss8YOP5eOB6ZNXJNUYtgL7sFb9O9a33BOjP3F0uPJIW34mA4nOQn0ArMWonRk58p8hElUW3uiKeuPxEah1lGkacacft1lO1LzJcyt8DslRH8nntHHrVmVd1FgrTe12QA9JoLN21RBiSFqbjyJbTLi090oUsAkfgTWLq4qUoQfDYoc5Prvq2+jplo/T+ndKH7msbQVHbZhnaE7SPPPJPJUrkqJJJNdKqKjBxjhIO5WaA/R1/eveoHH31reUnA3uK3KOSAOT+NElT2kYyvctq2pKYzDSQAgIGQOwAHAoLSYUANY6ljybhshOuhUELMhxs7eRg4Sc9wR39TUHTSe7wPuxYgn7/kQbzJuDRCX3XS+pRAyc9wfaq1LT7ZubWWPKpdWBC5Sha7zKuMIiNIkOqdJZO3lZJUB9STUY6ZU6r2q13ck6jlHJJQvTfUTTiJ6gk3WOA1JGMF1OOFfX/oasVaKnysjRnZEfTEOWyc282pTK0nLchv5VoP19Kqxoem90R3O+GH+m+pLkiOu2XxpmQy6nwypxAKHR6LHatSnU3YkAatwKnbc/oSQ3f8ATSlP2dPMmDu3LYGe6fVH6ijNvh8EcXugottktLkl+9QChi23JsvLQk4S25t+f8FZzj1zVXUxp1qFSlW+600/k07/AJBqTlGcZQ5TRHcRUtUlClbUJQAAj+UcetfCVRU9rR7fnLfPgc4cNx5RWwlz5j82e31q9oela7qs/S0lFz+Swvm+F9WVK+ro0I3rSSBfqHp/UyZsWRaLO9Mikj7VJZUlfhJ9SkHdj3xxXrn2c6B1rpdOUdVS2wbT5i//AOW/qcb1DW6evNOjK7+TX6jfqG93qFaTbtNxhN1HLb8Nkr4bZBHK1nyA/M16lpcvc1hGFXbS2x5BDpX8JVk0VcHL9qJSdR6tkuF96ZISClCzydo8v71qupKbKUKSp+7J0iW1ClobSjxdowEtjIHtiipkuchPAsDC0/72kxk443ECjRuyDseRNNyX5S/sTzciMD8wQEu8e6TzVuMXbDAtj/J6cWVEFuW9bWUuKICyyNuD6j0oxCyIy+JPoLbOqvSC7W1UFcu4xmFP29ZQC826E5G1XvjHv5021NWZCot0fdcHx5hxXob7seQ0tiQ0stuNOJ2qQoHCkkHsQcisSthgIBXZGwVYPORWJqHgtRHpyIVp7c1QU7MnbBwXAykjBzRFUI2Ga4slgDg8Gr1KSkROcV8g1OcScboj9P8AEK6N8FJjjbntjwPp5U6YF4ZLmhb14OwFWMc0WMgkUS3B1ElaB84z6UmwqyO8TUIQoFSgB71Xkrh07Dfd9TIU+AlYKayNTRc1gswmkbsagS00DvAHesv0JRLHqJizTWqEztSRWd4O3KwCf6Rn+9avT6d66uuAFeXw2JCuWoZMoqSlalLcVg88k9hXYwilJtmZKV1ggb4p9Qv6Ve07ai+C+/GcmOtJ/kBVsRn/ANK/zrN1Mt0kkTacYore7qSQp0OIUUrByFA4xVaytkHd8jfJnPTHS46srWfMnmkopLAnd5YZ9No7r75Syla3lr2oCP4iryx71idRbukgkFnBe97qTeL9pKxQ7olH26GlPjLbOQpW0Amuw0NX1qSdT71skKqlAKdB3RyHdPCUSkrWhQ/A5qzNpxTHgmmXOiXLw7P9qcOUNshaj6gJzVRJIsMA7ZanYlvn3BKUqeeUX3mlHcQhRJOB7Zz+FSlxYaKu7kJ9RAi0zW5UdKhEez3Odqv5k/rkexo1OO5Aqj2O5Gl0uJkNqQ2s/wBSeaM4XVwO4Jun2rE2KVIaeWUfaEDYVcAKHr9e1DnTurxCwnbDCm9OR7pCblMgBt4kEf0rHcVVcch28Aw3IAy2s8p7H2pttsEb9yQLvqBvQbViuUdfg2u5xNxDzmUeMnCXGxnvnIUB7n0rkeq9dn0bWUoV47qNRcpfFFp5b8xyr91+RtaXQR1lCcoO04v6NPj5PDEqNYBuZMskUeBbXdsxLRIwlKhkoz7kHj0Iryv7VfbOprqE9N0x2pS+Fys9z8pLsuVfl+x0/S+kRoyVTU/fWbdvb5sILFp9EiOhyQypCVq8RDG7kn39qzfsn9k1rdvUOoK9P+mP/l7v28Lv8uTdU6k6b9GhyuX++47ybqzDWI8dsSJOcBKR8if+te8Uo06EFSpRSS7JWX4HFycpvdJ3Zka9SYEpDipjipLZBDTBCUo9if8ALmrEZWywTjfBIo0ZZ7za275GhohvykBbqow2nPO724INE/h6VVbkrfIHvlB2I71HYJFiu64jziXDgLQsnAUk9jiqsqEoPagqnuVxbpyyXKY4W4YKucKUCEoH496NCm7kZSwPl86T3i5Wh0MXdMWcoHYsN70oPuCeauenjkBfwU96m2D4iOjc2XPtl3RfoiSXPGhxRvSB/wAmc4+hNZ7jq6LvFqSAS3LuQ9qf9ov1dnaWj2Zx62xJ8d4qXc2ohDywONikE7Qc9zj8KnHVykrNZAepJrkAdOfHp1esGr7feJeoTeo0VwqXbJLaEMPpIwUq2pz9DnirPqyfJKM5LLZGmsdau9SNfX/VL8Ni3u3ia5MVEjZ8NoqP8IJ7/XzOaztTLc7jxy7i21ktFJrBrZuHVwrZbDjOe+ax5OzCMwMDB470aLXciDmo2dicgce1X9NK7E0MrCflq9LkT8AHXQlQ6NOlCs+lM/Yi0EFn1A7AcSUnHnTppgleId23XiEoHir/AFohPf5Pbl1RQ0goaKs+9QY/qeAe/wC0eWHtwWpQz51VlubJqpYfouunrhHAS7tUO4zzQnG4VVELdPayXYNQQrg6tammnB4qU5JLZ4VgeZwTRKSdOSkhSlfkuZC0tFk2q33aHcIsiEFB8TUvDwn2FHKXEqPAx5juPOtV13Fk1TTRRP4iNbsa+6u3y4w3kyLeypMKK6g5SptobQoH0Ktx/GqsnuYOo7uxGxORTAjOc+9MxE3/AA125Mm9PvqGSwytxOfI8Jz+prNqpSr/ACQejyWRiKTEcbLyVLaA+dKTzk1apVJUXdFmUFJZOOpOv+ndFymIjTq5l2huNpU2GyNiCQSFq7cDnj1qxU1cUrgXaLsWCT8WNltuhUrly2JDL2NiGXQVqHBASM85NWHrKMKbqTeBWdyKOkXxl3G4dULgi9KTGjzVkwkE/KhscBk+WeMg+ZJHpWVpuouvUe/CfA90vhJN60PxJcdi52pzdaphO5CTww76fQ+X4iumpNxd+wCqlJWIPRPcYkbc4GTjJ7e1aM7NbkZ8bp7QsbvkW8W9CblHLMhobW5rSf4sfyrHn9ay5zjTk9jNCCco/EiZulGjf9r9Oz4a2HIbz6N8UunALyRkEexB59jVZVN7ZY2WRHWorZcLDLInRHoqge60EBX0PY1ZVmrgGmnYM4mmbz1B6dWJ2ywmLrO0/c5D6oj/AHcaW1j5B/MoKHA+orhPtf0qr1TRKGni3NPs0nZ82vzwsc+Df6Rqo6aruqPDXdd/3cYOn0WPcby/ZzJizG/EMiW2ykpcZVu5bV/ThXy44PGK8K6d0mrruoU4VqcoqPN/C5/Hj6nZV9TCjQl6c07+Pf8AwSxdbstK/sMTmSvhSk8bR6CvoCM1BKMfl8jjHG/Iym6xoLyoMV0vTDw662Nys/0p9PrR1NRx3BtNnVEhuK2hBSlhXknO5R/96Mmmsg37Em9PtcRo8IWea4GSG1uJU8sYIKh59uMn6VoUZK1itUTvcBdY9RIOptRvvQEFbQAZbefd2NhKc9gOTkkn8qHUqKUsDxjgVadui4clt5qW466Dy20MII9CM5IqUW1wM0iY7RcTcIDcqKtZCgQpl052qHl/r1q4pJoE1YFdTx7veo7T8CMyl08LQt3KFp/KldjZPlT8eHTKRoHqSi4mzPWuPdUlal+EQy46O+1XYnHJHfjNU6tNblJFSatKyKs5+amGH6y84rPrkoPIXwE5KfSsSoWEE1vkZbUk8kGsqrHNyVxWcYHlQk2IY9QM+KycDnvxV7TO0hNgwn5RitcYA66EqmUhHRDhTUWiLRt9pX5Kp8kdqOa3FKPJpWJJJG7ajiotDNCpt9bStyVFJ9qHYb2O7l1kKRtKs/WkkLk4OX65Ltq7aZ8n7uWvxFQ/GV4JV3zszjP4UdN2sETaEHakIz/XFIRg/CkMS50ava9NyWpiUeKjlDjecbkHvj37flXP6ut6Nbcg1J7ck13fXcibHQq0MKDyiEIS8kErWo4SMA+uPPzrLqdSnUqxp0ljvcO6l/uj7pHoNorT1qfvetr+rUeqn3t7lrgJ3NJUokqKnDgEDtkeZ4FdHKnG15vI8acVl5Y/a86o6a0l0+n2fSuiLZZHJrRjO3AoS9JLahhQCin5c57is/VVYQp+nShzi7CN2Vyqst3xJC1JVt/gAIOCOd1Z8FtS+pRly/oTboDr61EZb0zqKVxIaCm3XFfKvnACj5K47+ddR0zWVJQaqZs7E5WeGH4jR5UkIaUp1ladyHB/EPb3xXQutZc4Aqnd5EvUHqEx0uuWhrkhluekNy/ttrUsJEhshGxZByAQc4z71yfU8bZ0cSbv8/matCajibLNaS6qonotc+wpzBtcVue8Cnl11zBdx7JbwB9KuaepwlxbPzGlltk2mXbb1PududaakRJUVUhtDiApJQ4gqB59DmtYBaxHvUrqrP6W6Ab+4LaZ99uDCW4yWmx4UccBTznIAxwAPMnnhPOB1nqD6dp3Uisu+ey93j8P3e9o6Ea1S03hfn7f5Ic0FfbhBDst+DHYuMnKpD+7KnVEklR98k1430vTdSWrlqKU7xlzv557cs6nV1aEoqCXHgJJF1krZUiPIEdTnDj20qV74r06FOq1a6RgSnFdhbarYbdbXHIqXZG84VJYSN30GavwobYgJVLjbJsT9xYU5bb5HTIOQWrm04yc/wCNIWP0oipSf3ZfiRdRd0Rb1Ms/VLSNkk3ZrSzF5tbCSt6dbbh9sDScd1NpQlYHvjFJaeq8yf4AZ1UleKKxXT4jtWLcBiyI0YDt4bIP981GTUHZMo+tKR0tfxP9Rrc6lxi+JbWnkH7Og/5VB6ia4H3vyGVj+N3q9FcJRqFlSO5QqG3hXtwKaWuq0+AO+beWfRjpd1fgX/oVbtVSpUZ6dAtKZl3aZUkJSUoKl4GeOAcA/St/T1fVpRm/BYk9q3HTq70+018TfQmfbG3GZ9uu8ATbVPThRac272nEn1zj9RVpbZq/Zg5rcsHwrnwXbZPkw5AAfjuqZcA8lJJSf1BqhJWbRWvfI8WJOSKzNQyUAzt7e1IJPesKqw8cjtCVsd78Gqc8okO2Mo5ql3HQ1XMBSceVXKWGOCMkeG6tNbUXdXGSAH/Wa6MqmUhGcetIRlIRlIRuiosizvnjvQ7DHijkYpxCdXeiomjzypDmUhjPpSESZob5ben1rleoZmyceCW9KlKrlaUE/wD6gH8gT/lWJoob9ZH5r9ScOQ8fdVscVnnNdu1yw98AR1QL0ONDiONLbEhkyULWnAWgKKQR6jcFf+msfW/CoLyxSzgiZ1P75X/3D+gxVdPH77lVrIF6xkqfvzyc8NIS2Mewz/cmt3RR20E/N2RnyPui+tmr9B7EW25B2OjgRprSX28emDyPzrRUmhKbQpg6huWrZdxu91kqlSnlEkq/hSM8JSOyUjyArE18vjigkW5XbLD/AAo63j2GFeLaJRcmSHvEXEWSP93CEJOw5wc/MFD3H1otGezn98lmlZx23yWdtHVG8plx3WUoWwzFMJvxDhW0k7c8c4BrUjXeLIm4ruJNd9RY1n0zc7xeFFUOMwVrQnnCAQAAKq6inHURcaiun2JRm6fxeBo6S660z1iW2xpS7QVzFA7oktwMvII8ihXOD/UOPegU9HTjZRjYItQ55TD5eiruxelWa5SW7XLdQVR943tvewUMY9D6eYqytPZ24ZF1Lq5yNwtlm05cVXaLOhO2JLsiWbeSXltpwFkp7nAGePIEjvRHTjbOGiDqbc9hisPxM9IpaQy1rqda3TwFT2ypH470/wD+hUacqT4n+P8AwBdeLxwT30uvVkvznj2nUllvTS2ypEiE8gKPspG4ggjgitKMLK6GU1LhnyF6uosznVzWR04U/cH3vKMEND5A14hwE+3fHtisDVbVUltKsc8DEmMQ3x6VmOWQjWDrEBYCz5VCfxWQG1uDV68SYzD7LUp5ll4bXW23ClLg9FAHBH1qxT3LCFfB9C/2ZvWNd26aak0pc3dzenJHjxnHCMJjupUVI+gUFH8a6TSNqKgyxSzFt9j5m9Qp7F26g6lmxSPs0i5yXWinsUl1RGPwp6rTm2ipHhG9h7DntWRqQkWFcdZ2Jwax5IKhe28UISc8+dVmrsnfFx4TKy2Dk9qpOGRK413CRkHmrdOJL3Bqb8731NasMIXbkAOK6MqmfWkMZ50hzKQjPM0hzdsZVioy4IsUpQT5UJsY1WkjnFOncQmWOaKh0ZTjnhpCPRx5UhiU9FNbbc3n2rkNe71GGj90PPvVuxyrTKdVtaZlNqcPondgn8iazen/APyUx1jJONo085Jm2ovtbocqV4Oc8HBGQfwIruWrlhIjr4gZChrtEZf8MK1xIyQP5Qd7p/8A7BXPdTluqxiuyHas2Qqk7lpPbjcfxOf8qF2aK3cjm5PfaLhJcP8AM4T+tdPSjthGPsAeWJRRRg40ugNaeWrHKl1gat3rpFiH3TvCLaZrLjgSUIdSs58gFDNJuysRjbcXoi3dCQraoEeKnBB49q3YpItN3ZHvxH31X/ZDemUlWXiyycehdT39uKG3n5EajtBlM7Df7lpO/Q7xaJbkG5Q3A4xIbPKT/mCOCDwQakmZ8JuDwfTD4f8A4s9OfEDppjSus3W7NqRABZljjw3U/wALqFH+U/wnnjOFcYVVmMlUW1l2FRTJQaYW/qONJuUcKucBaLZeo+AUzYbv7tD6T2UNqgM+aVI/pNESs8/UJa+D5b61s7dh1JebW04l9qBNfiIdSeFpbcUgKH1Cc/jXNSjsqyinwzNeQZ+xNqcC9idx5zjmi+pJK1wSdx5gRkq4xx24qjVkXKaH6LbvETkis2dSzLijdCW5oSwjaBijUm5O4KSXDA+6SvDJAzityjC5VlZHlk6l6n0zZrlabRepVut1ySUSmY6tnipPcEjnkcd61ILZlcjqT27XwDbKfmAA+lJsjyEtmRtSOOay67uEiErKglIGKy5E1gUJO5OBQeGTFni/ugkdwKBbI5tFtT1ycASn5R3NWaMd7sh0hPeLI5GeSEpBHqBVypFUkSxYiPgGuiKRmMZpCPDinEZSHM86Q53ioBOTQ5sG+RwSnGP+lVmyZq40FA84P0qSYmrjY4nCiKtLgZGtOSPcZA5xSImyE5UBUbjEu6MaxBaGcdq4vXv42Wor4Rb1BdCLclsHgkAigdNV6lwc1gkXp58TabPo5qy3q3qkSoxQWLi2d3KP4FKR3CwPlJBwoAZAI56v1klZrISnWstsgQ17rBes7tMvqt+ZYCgHAAoAICE8DtwBxWBWl6uob+X5BW7x3IB5Kyy1JWDnYggfgKPBbnFeQHBHLo2rPr51064KyNaccP7S34Omo47FXNc5We7USLC+4bMRy6CkJ3FQIxSlK2RoOzuWttFwH3TEUhQUFMsqCh5/IK3KbvTT+X6Fh8gb17kur6TTHDgNOTIzHzHGVFSl8Dz4Qafm7IVsQKpOp+eoozWONnfciSW3mXFtPNnchxtRSpJ9QRyDQajayhoN3J3tfxO9QWtMmy/fKFI8H7Oiapj/AHptrOdqXM9s8jKSQexFAlrasY27+S16kmrEYTGC6VLJKs8kqPJrNjPOQbwNzg8MpTnj3qys5K6WR5tDOQCe1Ua77GhTXcJGiGmfTisp5ZdwkDV6fG4nPlWrQiV5WQC3V/xXcA10NGO1ZKMmrjYpGPrVq46Z2iNblZoc3ZDpXCi0NDbmsquwqQ9tpOBziqDZNIVRm9yCPegzdmSSHKIykrAJqrOTsMrB7pS3Ntw3nFgDA4zV7RKycmFEjUFN0vzbKUgoKgCKuVZbmosexWYjOK37lAzFK4jw9/WnuOeY8/KlcR5TjiyGMgetAmwbdmOCGVVWckTXsauJKeDxUk7iGp4ZWT6+VW48DI5pHNSY7PcUwx3jN73BwaHJ2QxMGkGx9nZFcVrX8TZopfChL1GeAWhAPZWcUbpccXKtQC2pGchJyfrW64+QCdwym4aYjs5wEBKfyHP9qw6eZORelxY42uKzPdDcnHgrBK9x45olacqavDkVOO92Ylnasstja+xRoUZ5QyFlLYWnP1PejU9HqdQ/UnNr62IucYfCkDUtmLqYIVBcaiPNjYIiwlCSM5ynHrmtSEqmkxVTaffn8Qb/AJixj2D1zSFwhaVtk1xttMN7LSHA4D8w7ggdq55aylPUzgvvLPBYdOWxPsNqC3DSvadysElWeKsu9S1yCtEm3pnfGrhoq0OF9tx5prwHEKdG4KQSkZGcjjBGe+a6CMlFWCR+KKdwH+IbXw1Cq3adihsRbe4qS8WznLyk7UpP+FO78VmnlUSW1AK0ruy7EFSRtdNSjwUZLJ2iLIUMZ/CoTQPgIra7l0A5rMqrAaLyO72EsqUapLLsFfcZHlBTg5HBq/FWQFRuwjs4+UVl13k0oIc57vhtHBxxVWlG7CzYDXyWopVhVdBp4IoVZAyoEnJ5rUKlzRSMnmpXJp2FURGO470ObDRCazo+TgVlV3kNEe2kBQrPk7BEOAYCGjVZyuxnhGn2nwFIVnz5p9m66IxYewrqj7rQWlj5hzzVyjK0Sx2CDplBTOvHirwTuplO8uSSVyoOe1dYZp6OaYYwUhHnfFOOYexphhbCO3+9Ankg3ZjshWcDFVGSvc5SuWyanDkZ5GRferqHXBrTjnoGTikIXwEAup+tAqPAO/xIlvSQwln864vW8s1HlA51FeUZ+PLJrV6ZFbDOrSd7AxZR4t1itnsp1IP51rV8U5NeAdJ/GkGV3cJeUcfyn9cCsKgsF6bGC/rPgNoGQkkqPv6Vo6ZZbB1HaKS7ghJ4crajwAieNDc42D5qFO3hkiTZbi/AaZ3q8NIBCNxxn1xXKQSu5WyW6jsJ5zYYgLKe5RiiU3umkwbdkMDaRkZAJ9a0mwKebM2eOBgCmjySbwMkrl8j3q/H7pWbvdjtboiScn+1U6s2KMcj5EjpQrI8qz5ybwG4N7i+r7KrHHFRpxW4dvAxJdLj6c1otWiRpZe4M7OkBCfpWFqOWacDa8uENmmoK7IzYC3hZOBjzroaCM2s8pDURgirYG9j1PI+tOTT7CxhACU1Xk8h0x/tAG0jyFZtbkKnyP8AGA3J471nTCoeVNAx8+1UU/iBydkDlzWW14B7VqUkmiMHdnOPd5DSA2lWEk9qI6Syw1yb+iTqlPoWeT3rJT21mkW6a+G5/9k=\"\n", + " }\n", + " },\n", + " \"params\": {\n", + " \"score_threshold\": \"0.8\"\n", + " }\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "TIqWQ1uzxXih" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "k6DeqQWixXih" + }, + "outputs": [], + "source": [ + "request = clients[\"prediction\"].predict(\n", + " name=model_id,\n", + " payload=payload,\n", + " params=params\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sIls-rRaxXih" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "1gfU3dZ6xXii", + "outputId": "fd3d5e94-e085-44f2-b196-1a438478dacc" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "2uqi9HpyxXii" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"payload\": [\n", + " {\n", + " \"annotationSpecId\": \"8907226159985459200\",\n", + " \"classification\": {\n", + " \"score\": 1.0\n", + " },\n", + " \"displayName\": \"daisy\"\n", + " }\n", + " ]\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "j14fFDBnxXii" + }, + "source": [ + "### [projects.locations.models.undeploy](https://cloud.google.com/automl/docs/reference/rest/v1/projects.locations.models/undeploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wrhN_FT4xXii" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "kR-o0CO7xXii" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].undeploy_model(\n", + " name=model_id\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "JF1qgjOfxXij" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "-VbCkwYyxXij" + }, + "outputs": [], + "source": [ + "result = request.result()\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "SpuhCYjSxXij" + }, + "source": [ + "#### Response\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ot9hRqTVxXij", + "outputId": "1d17e1c1-2be9-4f4f-a25b-404ee7465f15" + }, + "outputs": [], + "source": [ + "print(MessageToJson(result))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "4tj-vFT9xXij" + }, + "source": [ + "*Example output*:\n", + "\n", + "```\n", + "{}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "orEjC6NLxXij" + }, + "source": [ + "## Train and export an Edge model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "CLL_rGa4xXik" + }, + "source": [ + "### [projects.locations.models.create](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "mQBsYjcVxXik" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "twgvR-l2xXil", + "outputId": "82a0c11b-b3e0-45d8-b41d-c52e4975c68b" + }, + "outputs": [], + "source": [ + "# creating edge model for export\n", + "model_edge = {\n", + " \"display_name\": \"flowers_edge_\" + TIMESTAMP,\n", + " \"dataset_id\": dataset_short_id,\n", + " \"image_classification_model_metadata\":{\n", + " \"train_budget_milli_node_hours\": 8000,\n", + " \"model_type\": \"mobile-versatile-1\"\n", + " },\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " automl.CreateModelRequest(\n", + " parent=PARENT,\n", + " model=model_edge \n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "RRO7tHfhxXim" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"model\": {\n", + " \"displayName\": \"flowers_edge_20210226015151\",\n", + " \"datasetId\": \"ICN2833688305139187712\",\n", + " \"imageClassificationModelMetadata\": {\n", + " \"modelType\": \"mobile-versatile-1\",\n", + " \"trainBudgetMilliNodeHours\": \"8000\"\n", + " }\n", + " }\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "WpgXZxIbxXim" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "__Fnz5YPxXim" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].create_model(\n", + " parent=PARENT,\n", + " model=model_edge\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "N85ltvTexXim" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "808TqLvNxXim", + "outputId": "bbad5246-0b95-4f58-f39e-4483ecb0e6dc" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tlldfZhCxXin" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/ICN8566948201909714944\"\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "MfLa8cMoxXin" + }, + "outputs": [], + "source": [ + "model_edge_id = result.name\n", + " " + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "EK06N_YWxXin" + }, + "source": [ + "### [projects.locations.models.export](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models/export)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "AS54kVvbxXin", + "outputId": "f067c031-8ef1-4938-cc40-0e3c7ff8f65c" + }, + "outputs": [], + "source": [ + "output_config = {\n", + " \"model_format\": \"tflite\",\n", + " \"gcs_destination\": {\n", + " \"output_uri_prefix\": \"gs://\" + f\"{BUCKET_NAME}/export/\",\n", + " }\n", + "}\n", + "\n", + "\n", + "print(MessageToJson(\n", + " automl.ExportModelRequest(\n", + " name=model_edge_id,\n", + " output_config=output_config\n", + " ).__dict__[\"_pb\"])\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "fOH9ne7ixXin" + }, + "source": [ + "*Example output*:\n", + "\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/ICN8566948201909714944\",\n", + " \"outputConfig\": {\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210226015151/export/\"\n", + " },\n", + " \"modelFormat\": \"tflite\"\n", + " }\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "FyK175Y1xXin" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Tb-0zJeGxXin" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].export_model(\n", + " name=model_edge_id,\n", + " output_config=output_config\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "c7o6520bxXio" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Y8RN4-SvxXio", + "outputId": "31c811f0-7f4f-4735-e4ed-5e959e99f843" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bml24eNjxXio" + }, + "source": [ + "*Example output*:\n", + "\n", + "```\n", + "{}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "mbbVNOBRxXio", + "outputId": "98259422-edef-4807-8656-55b147440087" + }, + "outputs": [], + "source": [ + "model_export_dir = output_config[\"gcs_destination\"][\"output_uri_prefix\"]\n", + "\n", + "! gsutil ls -r $model_export_dir\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cBYOnQUDxXio" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210226015151/export/:\n", + "\n", + "gs://migration-ucaip-trainingaip-20210226015151/export/model-export/:\n", + "\n", + "gs://migration-ucaip-trainingaip-20210226015151/export/model-export/icn/:\n", + "\n", + "gs://migration-ucaip-trainingaip-20210226015151/export/model-export/icn/tflite-flowers_edge_20210226015151-2021-02-26T06:16:19.437101Z/:\n", + "gs://migration-ucaip-trainingaip-20210226015151/export/model-export/icn/tflite-flowers_edge_20210226015151-2021-02-26T06:16:19.437101Z/dict.txt\n", + "gs://migration-ucaip-trainingaip-20210226015151/export/model-export/icn/tflite-flowers_edge_20210226015151-2021-02-26T06:16:19.437101Z/model.tflite\n", + "gs://migration-ucaip-trainingaip-20210226015151/export/model-export/icn/tflite-flowers_edge_20210226015151-2021-02-26T06:16:19.437101Z/tflite_metadata.json\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup:migration,new" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "IRwi0SHBxXio", + "outputId": "7309e56d-ea43-4d70-da6d-c00de8fe6f5c", + "scrolled": true + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_model = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the AutoML fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset:\n", + " clients['automl'].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the AutoML fully qualified identifier for the model\n", + "try:\n", + " if delete_model:\n", + " clients['automl'].delete_model(name=model_id)\n", + "except Exception as e:\n", + " print(e)\n", + " \n", + "# Delete the model using the AutoML fully qualified identifier for the model\n", + "try:\n", + " if delete_model:\n", + " clients['automl'].delete_model(name=model_edge_id)\n", + "except Exception as e:\n", + " print(e) \n", + "\n", + "if delete_bucket and 'BUCKET_NAME' in globals():\n", + " ! gsutil rm -r gs://$BUCKET_NAME\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "BZDoIRo0UEhJ" + }, + "outputs": [], + "source": [] + } + ], + "metadata": { + "colab": { + "collapsed_sections": [ + "gcp_authenticate", + "bucket:batch_prediction", + "setup_vars", + "import_aip", + "aip_constants", + "5j4eOtdHxXif" + ], + "name": "[UJ.1 OLD] AutoML Vision Image Classification.ipynb", + "provenance": [] + }, + "environment": { + "name": "tf2-2-3-gpu.2-3.m55", + "type": "gcloud", + "uri": "gcr.io/deeplearning-platform-release/tf2-2-3-gpu.2-3:m55" + }, + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.7.8" + } + }, + "nbformat": 4, + "nbformat_minor": 4 +} diff --git a/notebooks/community/migration/UJ1 unified AutoML Vision Image Classification.ipynb b/notebooks/community/migration/UJ1 unified AutoML Vision Image Classification.ipynb new file mode 100644 index 000000000..c405f1593 --- /dev/null +++ b/notebooks/community/migration/UJ1 unified AutoML Vision Image Classification.ipynb @@ -0,0 +1,3274 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title:migration,new" + }, + "source": [ + "# Vertex SDK: AutoML image classification model\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest (preview) version of Vertex SDK.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "A_QVTkXr_i_r" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-aiplatform --user" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the Google *cloud-storage* library as well.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "65SBis8d_i_s" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-storage" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the Kernel\n", + "\n", + "Once you've installed the Vertex SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "GIwKc4pk_i_t" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "if not os.getenv(\"AUTORUN\") and False:\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU run-time\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your GCP project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a GCP project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebooks.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex AI. We recommend when possible, to choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You cannot use a Multi-Regional Storage bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see [Region support for Vertex AI services](https://cloud.google.com/vertex-ai/docs/general/locations)\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "6Xtp5tvK_i_y" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "d70An-Mg_i_2" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your GCP account\n", + "\n", + "**If you are using Google Cloud Notebooks**, your environment is already\n", + "authenticated. Skip this step.\n", + "\n", + "*Note: If you are on an Vertex notebook and run the cell, the cell knows to skip executing the authentication steps.*\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "chybg3Ap_i_2" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your Google Cloud account. This provides access\n", + "# to your Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Vertex, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this tutorial in a notebook locally, replace the string\n", + " # below with the path to your service account key and run this cell to\n", + " # authenticate your Google Cloud account.\n", + " else:\n", + " %env GOOGLE_APPLICATION_CREDENTIALS your_path_to_credentials.json\n", + "\n", + " # Log in to your account on Google Cloud\n", + " ! gcloud auth login" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n", + " BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "qPPrwWpO_i_6" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "fn744B7x_i_7" + }, + "outputs": [], + "source": [ + "! gsutil ls -al gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip" + }, + "source": [ + "#### Import Vertex SDK\n", + "\n", + "Import the Vertex SDK into our Python environment.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "97-XQPkv_i_7" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex AI constants\n", + "\n", + "Setup up the following constants for Vertex AI:\n", + "\n", + "- `API_ENDPOINT`: The Vertex AI API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `API_PREDICT_ENDPOINT`: The Vertex AI API service endpoint for prediction.\n", + "- `PARENT`: The Vertex AI location root path for dataset, model and endpoint resources.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "kBBYqHEd_i_8" + }, + "outputs": [], + "source": [ + "# API Endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex AI location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants:automl" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Next, setup constants unique to AutoML image classification datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the managed dataset service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the managed dataset service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the managed pipelines service the task (e.g., classification) to train the model for.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:automl,icn" + }, + "outputs": [], + "source": [ + "# Image Dataset type\n", + "IMAGE_SCHEMA = \"google-cloud-aiplatform/schema/dataset/metadata/image_1.0.0.yaml\"\n", + "# Image Labeling type\n", + "IMPORT_SCHEMA_IMAGE_CLASSIFICATION = \"gs://google-cloud-aiplatform/schema/dataset/ioformat/image_classification_single_label_io_format_1.0.0.yaml\"\n", + "# Image Training task\n", + "TRAINING_IMAGE_CLASSIFICATION_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_image_classification_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients" + }, + "source": [ + "## Clients\n", + "\n", + "The Vertex SDK works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the server (Vertex).\n", + "\n", + "You will use several clients in this tutorial, so set them all up upfront.\n", + "\n", + "- Dataset Service for managed datasets.\n", + "- Model Service for managed models.\n", + "- Pipeline Service for training.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving. *Note*: Prediction has a different service endpoint.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "i2p2VYUz_i_-" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "clients[\"job\"] = create_job_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:flowers,csv,icn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = (\n", + " \"gs://cloud-samples-data/vision/automl_classification/flowers/all_data_v2.csv\"\n", + ")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cBgFrDD1_i__" + }, + "outputs": [], + "source": [ + "! gsutil cat $IMPORT_FILE | head -n 10" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_create:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://cloud-ml-data/img/flower_photos/daisy/100080576_f52e8ee070_n.jpg,daisy\n", + "gs://cloud-ml-data/img/flower_photos/daisy/10140303196_b88d3d6cec.jpg,daisy\n", + "gs://cloud-ml-data/img/flower_photos/daisy/10172379554_b296050f82_n.jpg,daisy\n", + "gs://cloud-ml-data/img/flower_photos/daisy/10172567486_2748826a8b.jpg,daisy\n", + "gs://cloud-ml-data/img/flower_photos/daisy/10172636503_21bededa75_n.jpg,daisy\n", + "gs://cloud-ml-data/img/flower_photos/daisy/102841525_bd6628ae3c.jpg,daisy\n", + "gs://cloud-ml-data/img/flower_photos/daisy/1031799732_e7f4008c03.jpg,daisy\n", + "gs://cloud-ml-data/img/flower_photos/daisy/10391248763_1d16681106_n.jpg,daisy\n", + "gs://cloud-ml-data/img/flower_photos/daisy/10437754174_22ec990b77_m.jpg,daisy\n", + "gs://cloud-ml-data/img/flower_photos/daisy/10437770546_8bb6f7bdd3_m.jpg,daisy\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_a_dataset:migration" + }, + "source": [ + "## Create a dataset\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_create:migration,new" + }, + "source": [ + "### [projects.locations.datasets.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.datasets/create)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "request:migration" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "RsM4amS9_jAB" + }, + "outputs": [], + "source": [ + "DATA_SCHEMA = IMAGE_SCHEMA\n", + "\n", + "dataset = {\n", + " \"display_name\": \"flowers_\" + TIMESTAMP,\n", + " \"metadata_schema_uri\": \"gs://\" + DATA_SCHEMA,\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateDatasetRequest(\n", + " parent=PARENT,\n", + " dataset=dataset,\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"dataset\": {\n", + " \"displayName\": \"flowers_20210226014942\",\n", + " \"metadataSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/metadata/image_1.0.0.yaml\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "call:migration" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "datasets_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"dataset\"].create_dataset(\n", + " parent=PARENT,\n", + " dataset=dataset,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "response:migration" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,response" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_create:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/datasets/3094342379910463488\",\n", + " \"displayName\": \"flowers_20210226014942\",\n", + " \"metadataSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/metadata/image_1.0.0.yaml\",\n", + " \"labels\": {\n", + " \"aiplatform.googleapis.com/dataset_metadata_schema\": \"IMAGE\"\n", + " },\n", + " \"metadata\": {\n", + " \"dataItemSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/dataitem/image_1.0.0.yaml\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new" + }, + "source": [ + "### [projects.locations.datasets.import](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.datasets/import)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "oJtL08_Q_jAF" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "outputs": [], + "source": [ + "LABEL_SCHEMA = IMPORT_SCHEMA_IMAGE_CLASSIFICATION\n", + "\n", + "import_config = {\n", + " \"gcs_source\": {\n", + " \"uris\": [IMPORT_FILE],\n", + " },\n", + " \"import_schema_uri\": LABEL_SCHEMA,\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.ImportDataRequest(\n", + " name=dataset_short_id,\n", + " import_configs=[import_config],\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gC5MZDJc_jAG" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"3094342379910463488\",\n", + " \"importConfigs\": [\n", + " {\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://cloud-samples-data/vision/automl_classification/flowers/all_data_v2.csv\"\n", + " ]\n", + " },\n", + " \"importSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/ioformat/image_classification_single_label_io_format_1.0.0.yaml\"\n", + " }\n", + " ]\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "rnhDF5vW_jAG" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "datasets_import:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"dataset\"].import_data(\n", + " name=dataset_id,\n", + " import_configs=[import_config],\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "zQoFJ2K0_jAH" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "K7BjtQ3e_jAH" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_a_model:migration" + }, + "source": [ + "## Train a model\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "trainingpipelines_create:migration,new" + }, + "source": [ + "### [projects.locations.trainingPipelines.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.trainingPipelines/create)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "m5igPySU_jAJ" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_create:migration,new,request,icn" + }, + "outputs": [], + "source": [ + "TRAINING_SCHEMA = TRAINING_IMAGE_CLASSIFICATION_SCHEMA\n", + "\n", + "task = Value(\n", + " struct_value=Struct(\n", + " fields={\n", + " \"multi_label\": Value(bool_value=False),\n", + " \"model_type\": Value(string_value=\"CLOUD\"),\n", + " \"budget_milli_node_hours\": Value(number_value=8000),\n", + " \"disable_early_stopping\": Value(bool_value=False),\n", + " }\n", + " )\n", + ")\n", + "\n", + "training_pipeline = {\n", + " \"display_name\": \"flowers_\" + TIMESTAMP,\n", + " \"input_data_config\": {\n", + " \"dataset_id\": dataset_short_id,\n", + " },\n", + " \"model_to_upload\": {\n", + " \"display_name\": \"flowers_\" + TIMESTAMP,\n", + " },\n", + " \"training_task_definition\": TRAINING_SCHEMA,\n", + " \"training_task_inputs\": task,\n", + "}\n", + "\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateTrainingPipelineRequest(\n", + " parent=PARENT,\n", + " training_pipeline=training_pipeline,\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "7IQ6Jp8E_jAJ" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"trainingPipeline\": {\n", + " \"displayName\": \"flowers_20210226014942\",\n", + " \"inputDataConfig\": {\n", + " \"datasetId\": \"3094342379910463488\"\n", + " },\n", + " \"trainingTaskDefinition\": \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_image_classification_1.0.0.yaml\",\n", + " \"trainingTaskInputs\": {\n", + " \"model_type\": \"CLOUD\",\n", + " \"budget_milli_node_hours\": 8000.0,\n", + " \"multi_label\": false,\n", + " \"disable_early_stopping\": false\n", + " },\n", + " \"modelToUpload\": {\n", + " \"displayName\": \"flowers_20210226014942\"\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "xhuR86RL_jAK" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT,\n", + " training_pipeline=training_pipeline,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "S8zP7wju_jAL" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,request" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "trainingpipelines_create:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/trainingPipelines/1112934465727889408\",\n", + " \"displayName\": \"flowers_20210226014942\",\n", + " \"inputDataConfig\": {\n", + " \"datasetId\": \"3094342379910463488\"\n", + " },\n", + " \"trainingTaskDefinition\": \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_image_classification_1.0.0.yaml\",\n", + " \"trainingTaskInputs\": {\n", + " \"budgetMilliNodeHours\": \"8000\",\n", + " \"modelType\": \"CLOUD\"\n", + " },\n", + " \"modelToUpload\": {\n", + " \"displayName\": \"flowers_20210226014942\"\n", + " },\n", + " \"state\": \"PIPELINE_STATE_PENDING\",\n", + " \"createTime\": \"2021-02-26T02:11:57.377842Z\",\n", + " \"updateTime\": \"2021-02-26T02:11:57.377842Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "training_pipeline_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the training pipeline\n", + "training_pipeline_id = request.name\n", + "# The short numeric ID for the training pipeline\n", + "training_pipeline_short_id = training_pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(training_pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "trainingpipelines_get:migration,new" + }, + "source": [ + "### [projects.locations.trainingPipelines.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.trainingPipelines/get)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "A3jRv70o_jAN" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_get:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"pipeline\"].get_training_pipeline(\n", + " name=training_pipeline_id,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "XC5I2xxt_jAN" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "yXZnQR1t_jAO" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "trainingpipelines_get:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/trainingPipelines/1112934465727889408\",\n", + " \"displayName\": \"flowers_20210226014942\",\n", + " \"inputDataConfig\": {\n", + " \"datasetId\": \"3094342379910463488\"\n", + " },\n", + " \"trainingTaskDefinition\": \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_image_classification_1.0.0.yaml\",\n", + " \"trainingTaskInputs\": {\n", + " \"budgetMilliNodeHours\": \"8000\",\n", + " \"modelType\": \"CLOUD\"\n", + " },\n", + " \"modelToUpload\": {\n", + " \"displayName\": \"flowers_20210226014942\"\n", + " },\n", + " \"state\": \"PIPELINE_STATE_PENDING\",\n", + " \"createTime\": \"2021-02-26T02:11:57.377842Z\",\n", + " \"updateTime\": \"2021-02-26T02:11:57.377842Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=training_pipeline_id)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " break\n", + " else:\n", + " model_id = response.model_to_upload.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:migration" + }, + "source": [ + "## Evaluate the model\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "models_evaluations_list:migration,new" + }, + "source": [ + "### [projects.locations.models.evaluations.list](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.models.evaluations/list)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Ngn6qqVy_jAQ" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "models_evaluations_list:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"model\"].list_model_evaluations(\n", + " parent=model_id,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "F0ryqI3F_jAQ" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "AxEY7WFHj2YC" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "model_evaluations = [json.loads(MessageToJson(me.__dict__[\"_pb\"])) for me in request]\n", + "# The evaluation slice\n", + "evaluation_slice = request.model_evaluations[0].name\n", + "\n", + "print(json.dumps(model_evaluations, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "models_evaluations_list:migration,new,response,icn" + }, + "source": [ + "*Example output*\n", + "```\n", + "[\n", + " {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/6656478578927992832/evaluations/8656839874550169600\",\n", + " \"metricsSchemaUri\": \"gs://google-cloud-aiplatform/schema/modelevaluation/classification_metrics_1.0.0.yaml\",\n", + " \"metrics\": {\n", + " \"confidenceMetrics\": [\n", + " {\n", + " \"precision\": 0.2,\n", + " \"recall\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.05,\n", + " \"recall\": 0.98092645,\n", + " \"precision\": 0.8910891\n", + " },\n", + " {\n", + " \"recall\": 0.97275203,\n", + " \"confidenceThreshold\": 0.1,\n", + " \"precision\": 0.92248064\n", + " },\n", + " {\n", + " \"recall\": 0.97002727,\n", + " \"confidenceThreshold\": 0.15,\n", + " \"precision\": 0.9295039\n", + " },\n", + " {\n", + " \"precision\": 0.93421054,\n", + " \"confidenceThreshold\": 0.2,\n", + " \"recall\": 0.96730244\n", + " },\n", + " {\n", + " \"precision\": 0.9465241,\n", + " \"recall\": 0.9645777,\n", + " \"confidenceThreshold\": 0.25\n", + " },\n", + " {\n", + " \"recall\": 0.9645777,\n", + " \"precision\": 0.9516129,\n", + " \"confidenceThreshold\": 0.3\n", + " },\n", + " {\n", + " \"precision\": 0.9567568,\n", + " \"recall\": 0.9645777,\n", + " \"confidenceThreshold\": 0.35\n", + " },\n", + " {\n", + " \"precision\": 0.9592391,\n", + " \"recall\": 0.96185285,\n", + " \"confidenceThreshold\": 0.4\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.45,\n", + " \"precision\": 0.96185285,\n", + " \"recall\": 0.96185285\n", + " },\n", + " {\n", + " \"precision\": 0.96185285,\n", + " \"recall\": 0.96185285,\n", + " \"confidenceThreshold\": 0.5\n", + " },\n", + " {\n", + " \"recall\": 0.96185285,\n", + " \"confidenceThreshold\": 0.55,\n", + " \"precision\": 0.9644809\n", + " },\n", + " {\n", + " \"recall\": 0.95640326,\n", + " \"confidenceThreshold\": 0.6,\n", + " \"precision\": 0.96428573\n", + " },\n", + " {\n", + " \"precision\": 0.96694213,\n", + " \"confidenceThreshold\": 0.65,\n", + " \"recall\": 0.95640326\n", + " },\n", + " {\n", + " \"recall\": 0.9536785,\n", + " \"confidenceThreshold\": 0.7,\n", + " \"precision\": 0.9695291\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.75,\n", + " \"precision\": 0.9719888,\n", + " \"recall\": 0.94550407\n", + " },\n", + " {\n", + " \"precision\": 0.97720796,\n", + " \"confidenceThreshold\": 0.8,\n", + " \"recall\": 0.9346049\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.85,\n", + " \"recall\": 0.9318801,\n", + " \"precision\": 0.9771429\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.875,\n", + " \"recall\": 0.9291553,\n", + " \"precision\": 0.97988504\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9,\n", + " \"precision\": 0.98255813,\n", + " \"recall\": 0.92098093\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.91,\n", + " \"precision\": 0.9825073,\n", + " \"recall\": 0.9182561\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.92,\n", + " \"recall\": 0.91553134,\n", + " \"precision\": 0.9882353\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.93,\n", + " \"recall\": 0.9128065,\n", + " \"precision\": 0.9882006\n", + " },\n", + " {\n", + " \"precision\": 0.98813057,\n", + " \"confidenceThreshold\": 0.94,\n", + " \"recall\": 0.907357\n", + " },\n", + " {\n", + " \"precision\": 0.990991,\n", + " \"recall\": 0.89918256,\n", + " \"confidenceThreshold\": 0.95\n", + " },\n", + " {\n", + " \"recall\": 0.8855586,\n", + " \"precision\": 0.9938838,\n", + " \"confidenceThreshold\": 0.96\n", + " },\n", + " {\n", + " \"precision\": 0.99380803,\n", + " \"recall\": 0.8746594,\n", + " \"confidenceThreshold\": 0.97\n", + " },\n", + " {\n", + " \"recall\": 0.8692098,\n", + " \"precision\": 0.99376947,\n", + " \"confidenceThreshold\": 0.98\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99,\n", + " \"precision\": 0.9968254,\n", + " \"recall\": 0.8555858\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.995,\n", + " \"recall\": 0.8310627,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"recall\": 0.8256131,\n", + " \"precision\": 1.0,\n", + " \"confidenceThreshold\": 0.996\n", + " },\n", + " {\n", + " \"recall\": 0.8092643,\n", + " \"confidenceThreshold\": 0.997,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.998,\n", + " \"precision\": 1.0,\n", + " \"recall\": 0.79019076\n", + " },\n", + " {\n", + " \"precision\": 1.0,\n", + " \"recall\": 0.76021796,\n", + " \"confidenceThreshold\": 0.999\n", + " },\n", + " {\n", + " \"precision\": 1.0,\n", + " \"confidenceThreshold\": 1.0,\n", + " \"recall\": 0.22888283\n", + " }\n", + " ],\n", + " \"confusionMatrix\": {\n", + " \"rows\": [\n", + " [\n", + " 80.0,\n", + " 0.0,\n", + " 0.0,\n", + " 0.0,\n", + " 0.0\n", + " ],\n", + " [\n", + " 3.0,\n", + " 85.0,\n", + " 0.0,\n", + " 2.0,\n", + " 0.0\n", + " ],\n", + " [\n", + " 0.0,\n", + " 1.0,\n", + " 67.0,\n", + " 1.0,\n", + " 1.0\n", + " ],\n", + " [\n", + " 1.0,\n", + " 1.0,\n", + " 1.0,\n", + " 60.0,\n", + " 0.0\n", + " ],\n", + " [\n", + " 3.0,\n", + " 0.0,\n", + " 0.0,\n", + " 0.0,\n", + " 61.0\n", + " ]\n", + " ],\n", + " \"annotationSpecs\": [\n", + " {\n", + " \"displayName\": \"tulips\",\n", + " \"id\": \"521556639170428928\"\n", + " },\n", + " {\n", + " \"displayName\": \"dandelion\",\n", + " \"id\": \"1674478143777275904\"\n", + " },\n", + " {\n", + " \"displayName\": \"sunflowers\",\n", + " \"id\": \"2827399648384122880\"\n", + " },\n", + " {\n", + " \"displayName\": \"daisy\",\n", + " \"id\": \"5133242657597816832\"\n", + " },\n", + " {\n", + " \"id\": \"7439085666811510784\",\n", + " \"displayName\": \"roses\"\n", + " }\n", + " ]\n", + " },\n", + " \"logLoss\": 0.04900711,\n", + " \"auPrc\": 0.99361706\n", + " },\n", + " \"createTime\": \"2021-02-26T02:36:30.247855Z\",\n", + " \"sliceDimensions\": [\n", + " \"annotationSpec\"\n", + " ]\n", + " }\n", + "]\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "models_evaluations_get:migration,new" + }, + "source": [ + "### [projects.locations.models.evaluations.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.models.evaluations/get)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "_NXujm2U_jAR" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "models_evaluations_get:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"model\"].get_model_evaluation(\n", + " name=evaluation_slice,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0RLTdCfj_jAS" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "HB2Uwhnq_jAS" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "models_evaluations_get:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/6656478578927992832/evaluations/8656839874550169600\",\n", + " \"metricsSchemaUri\": \"gs://google-cloud-aiplatform/schema/modelevaluation/classification_metrics_1.0.0.yaml\",\n", + " \"metrics\": {\n", + " \"logLoss\": 0.04900711,\n", + " \"confusionMatrix\": {\n", + " \"annotationSpecs\": [\n", + " {\n", + " \"displayName\": \"tulips\",\n", + " \"id\": \"521556639170428928\"\n", + " },\n", + " {\n", + " \"displayName\": \"dandelion\",\n", + " \"id\": \"1674478143777275904\"\n", + " },\n", + " {\n", + " \"displayName\": \"sunflowers\",\n", + " \"id\": \"2827399648384122880\"\n", + " },\n", + " {\n", + " \"displayName\": \"daisy\",\n", + " \"id\": \"5133242657597816832\"\n", + " },\n", + " {\n", + " \"id\": \"7439085666811510784\",\n", + " \"displayName\": \"roses\"\n", + " }\n", + " ],\n", + " \"rows\": [\n", + " [\n", + " 80.0,\n", + " 0.0,\n", + " 0.0,\n", + " 0.0,\n", + " 0.0\n", + " ],\n", + " [\n", + " 3.0,\n", + " 85.0,\n", + " 0.0,\n", + " 2.0,\n", + " 0.0\n", + " ],\n", + " [\n", + " 0.0,\n", + " 1.0,\n", + " 67.0,\n", + " 1.0,\n", + " 1.0\n", + " ],\n", + " [\n", + " 1.0,\n", + " 1.0,\n", + " 1.0,\n", + " 60.0,\n", + " 0.0\n", + " ],\n", + " [\n", + " 3.0,\n", + " 0.0,\n", + " 0.0,\n", + " 0.0,\n", + " 61.0\n", + " ]\n", + " ]\n", + " },\n", + " \"auPrc\": 0.99361706,\n", + " \"confidenceMetrics\": [\n", + " {\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.2\n", + " },\n", + " {\n", + " \"precision\": 0.8910891,\n", + " \"confidenceThreshold\": 0.05,\n", + " \"recall\": 0.98092645\n", + " },\n", + " {\n", + " \"recall\": 0.97275203,\n", + " \"precision\": 0.92248064,\n", + " \"confidenceThreshold\": 0.1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.15,\n", + " \"precision\": 0.9295039,\n", + " \"recall\": 0.97002727\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.2,\n", + " \"precision\": 0.93421054,\n", + " \"recall\": 0.96730244\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.25,\n", + " \"recall\": 0.9645777,\n", + " \"precision\": 0.9465241\n", + " },\n", + " {\n", + " \"precision\": 0.9516129,\n", + " \"recall\": 0.9645777,\n", + " \"confidenceThreshold\": 0.3\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.35,\n", + " \"precision\": 0.9567568,\n", + " \"recall\": 0.9645777\n", + " },\n", + " {\n", + " \"precision\": 0.9592391,\n", + " \"recall\": 0.96185285,\n", + " \"confidenceThreshold\": 0.4\n", + " },\n", + " {\n", + " \"recall\": 0.96185285,\n", + " \"precision\": 0.96185285,\n", + " \"confidenceThreshold\": 0.45\n", + " },\n", + " {\n", + " \"precision\": 0.96185285,\n", + " \"recall\": 0.96185285,\n", + " \"confidenceThreshold\": 0.5\n", + " },\n", + " {\n", + " \"precision\": 0.9644809,\n", + " \"recall\": 0.96185285,\n", + " \"confidenceThreshold\": 0.55\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6,\n", + " \"recall\": 0.95640326,\n", + " \"precision\": 0.96428573\n", + " },\n", + " {\n", + " \"recall\": 0.95640326,\n", + " \"precision\": 0.96694213,\n", + " \"confidenceThreshold\": 0.65\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.7,\n", + " \"precision\": 0.9695291,\n", + " \"recall\": 0.9536785\n", + " },\n", + " {\n", + " \"recall\": 0.94550407,\n", + " \"confidenceThreshold\": 0.75,\n", + " \"precision\": 0.9719888\n", + " },\n", + " {\n", + " \"recall\": 0.9346049,\n", + " \"precision\": 0.97720796,\n", + " \"confidenceThreshold\": 0.8\n", + " },\n", + " {\n", + " \"precision\": 0.9771429,\n", + " \"confidenceThreshold\": 0.85,\n", + " \"recall\": 0.9318801\n", + " },\n", + " {\n", + " \"precision\": 0.97988504,\n", + " \"confidenceThreshold\": 0.875,\n", + " \"recall\": 0.9291553\n", + " },\n", + " {\n", + " \"recall\": 0.92098093,\n", + " \"confidenceThreshold\": 0.9,\n", + " \"precision\": 0.98255813\n", + " },\n", + " {\n", + " \"recall\": 0.9182561,\n", + " \"confidenceThreshold\": 0.91,\n", + " \"precision\": 0.9825073\n", + " },\n", + " {\n", + " \"precision\": 0.9882353,\n", + " \"confidenceThreshold\": 0.92,\n", + " \"recall\": 0.91553134\n", + " },\n", + " {\n", + " \"precision\": 0.9882006,\n", + " \"confidenceThreshold\": 0.93,\n", + " \"recall\": 0.9128065\n", + " },\n", + " {\n", + " \"precision\": 0.98813057,\n", + " \"recall\": 0.907357,\n", + " \"confidenceThreshold\": 0.94\n", + " },\n", + " {\n", + " \"precision\": 0.990991,\n", + " \"confidenceThreshold\": 0.95,\n", + " \"recall\": 0.89918256\n", + " },\n", + " {\n", + " \"precision\": 0.9938838,\n", + " \"confidenceThreshold\": 0.96,\n", + " \"recall\": 0.8855586\n", + " },\n", + " {\n", + " \"recall\": 0.8746594,\n", + " \"precision\": 0.99380803,\n", + " \"confidenceThreshold\": 0.97\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98,\n", + " \"precision\": 0.99376947,\n", + " \"recall\": 0.8692098\n", + " },\n", + " {\n", + " \"precision\": 0.9968254,\n", + " \"confidenceThreshold\": 0.99,\n", + " \"recall\": 0.8555858\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.995,\n", + " \"precision\": 1.0,\n", + " \"recall\": 0.8310627\n", + " },\n", + " {\n", + " \"precision\": 1.0,\n", + " \"recall\": 0.8256131,\n", + " \"confidenceThreshold\": 0.996\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.997,\n", + " \"precision\": 1.0,\n", + " \"recall\": 0.8092643\n", + " },\n", + " {\n", + " \"recall\": 0.79019076,\n", + " \"precision\": 1.0,\n", + " \"confidenceThreshold\": 0.998\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.999,\n", + " \"recall\": 0.76021796,\n", + " \"precision\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 1.0,\n", + " \"precision\": 1.0,\n", + " \"recall\": 0.22888283\n", + " }\n", + " ]\n", + " },\n", + " \"createTime\": \"2021-02-26T02:36:30.247855Z\",\n", + " \"sliceDimensions\": [\n", + " \"annotationSpec\"\n", + " ]\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_predictions:migration" + }, + "source": [ + "## Make batch predictions\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_prediction_file:migration,new" + }, + "source": [ + "### Make a batch prediction file\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_items:automl,icn,csv" + }, + "outputs": [], + "source": [ + "test_items = !gsutil cat $IMPORT_FILE | head -n2\n", + "\n", + "if len(str(test_items[0]).split(\",\")) == 3:\n", + " _, test_item_1, test_label_1 = str(test_items[0]).split(\",\")\n", + " _, test_item_2, test_label_2 = str(test_items[1]).split(\",\")\n", + "else:\n", + " test_item_1, test_label_1 = str(test_items[0]).split(\",\")\n", + " test_item_2, test_label_2 = str(test_items[1]).split(\",\")\n", + "\n", + "print(test_item_1, test_label_1)\n", + "print(test_item_2, test_label_2)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "rQkHuZbk_jAV" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://cloud-ml-data/img/flower_photos/daisy/100080576_f52e8ee070_n.jpg daisy\n", + "gs://cloud-ml-data/img/flower_photos/daisy/10140303196_b88d3d6cec.jpg daisy\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copy_test_items:automl,image,batch_prediction" + }, + "outputs": [], + "source": [ + "file_1 = test_item_1.split(\"/\")[-1]\n", + "file_2 = test_item_2.split(\"/\")[-1]\n", + "\n", + "! gsutil cp $test_item_1 gs://$BUCKET_NAME/$file_1\n", + "! gsutil cp $test_item_2 gs://$BUCKET_NAME/$file_2\n", + "\n", + "test_item_1 = \"gs://\" + BUCKET_NAME + \"/\" + file_1\n", + "test_item_2 = \"gs://\" + BUCKET_NAME + \"/\" + file_2" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:automl,image" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "Let's now make a batch input file, which you will store in your local Cloud Storage bucket. The batch input file can be either CSV or JSONL. You will use JSONL in this tutorial. For JSONL file, you make one dictionary entry per line for each data item (instance). The dictionary contains the key/value pairs:\n", + "\n", + "- `content`: The Cloud Storage path to the image.\n", + "- `mime_type`: The content type. In our example, it is an `jpeg` file.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bxjbyhI3_jAW" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "import tensorflow as tf\n", + "\n", + "gcs_input_uri = \"gs://\" + BUCKET_NAME + \"/test.jsonl\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " data = {\"content\": test_item_1, \"mime_type\": \"image/jpeg\"}\n", + " f.write(json.dumps(data) + \"\\n\")\n", + " data = {\"content\": test_item_2, \"mime_type\": \"image/jpeg\"}\n", + " f.write(json.dumps(data) + \"\\n\")\n", + "\n", + "print(gcs_input_uri)\n", + "!gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "MC8D11Xe_jAW" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210226014942/test.jsonl\n", + "{\"content\": \"gs://migration-ucaip-trainingaip-20210226014942/100080576_f52e8ee070_n.jpg\", \"mime_type\": \"image/jpeg\"}\n", + "{\"content\": \"gs://migration-ucaip-trainingaip-20210226014942/10140303196_b88d3d6cec.jpg\", \"mime_type\": \"image/jpeg\"}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_create:migration,new" + }, + "source": [ + "### [projects.locations.batchPredictionJobs.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.batchPredictionJobs/create)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "htIpycBi_jAX" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_create:migration,new,request,icn" + }, + "outputs": [], + "source": [ + "parameters = {\"confidenceThreshold\": 0.5, \"maxPredictions\": 2}\n", + "\n", + "batch_prediction_job = {\n", + " \"display_name\": \"flowers_\" + TIMESTAMP,\n", + " \"model\": model_id,\n", + " \"input_config\": {\n", + " \"instances_format\": \"jsonl\",\n", + " \"gcs_source\": {\n", + " \"uris\": [gcs_input_uri],\n", + " },\n", + " },\n", + " \"model_parameters\": json_format.ParseDict(parameters, Value()),\n", + " \"output_config\": {\n", + " \"predictions_format\": \"jsonl\",\n", + " \"gcs_destination\": {\n", + " \"output_uri_prefix\": \"gs://\" + f\"{BUCKET_NAME}/batch_output/\",\n", + " },\n", + " },\n", + " \"dedicated_resources\": {\n", + " \"machine_spec\": {\n", + " \"machine_type\": \"n1-standard-2\",\n", + " \"accelerator_type\": 0,\n", + " },\n", + " \"starting_replica_count\": 1,\n", + " \"max_replica_count\": 1,\n", + " },\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateBatchPredictionJobRequest(\n", + " parent=PARENT,\n", + " batch_prediction_job=batch_prediction_job,\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cXThRLYf_jAX" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"batchPredictionJob\": {\n", + " \"displayName\": \"flowers_20210226014942\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/6656478578927992832\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210226014942/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"modelParameters\": {\n", + " \"confidenceThreshold\": 0.5,\n", + " \"maxPredictions\": 2.0\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210226014942/batch_output/\"\n", + " }\n", + " },\n", + " \"dedicatedResources\": {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-2\"\n", + " },\n", + " \"startingReplicaCount\": 1,\n", + " \"maxReplicaCount\": 1\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "8QO3y-36_jAY" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].create_batch_prediction_job(\n", + " parent=PARENT,\n", + " batch_prediction_job=batch_prediction_job,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "DmClxRYK_jAY" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "118r_5MI_jAY" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_create:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/batchPredictionJobs/7156765165659095040\",\n", + " \"displayName\": \"flowers_20210226014942\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/6656478578927992832\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210226014942/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"modelParameters\": {\n", + " \"maxPredictions\": 2.0,\n", + " \"confidenceThreshold\": 0.5\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210226014942/batch_output/\"\n", + " }\n", + " },\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"completionStats\": {\n", + " \"incompleteCount\": \"-1\"\n", + " },\n", + " \"createTime\": \"2021-02-26T02:36:52.483588Z\",\n", + " \"updateTime\": \"2021-02-26T02:36:52.483588Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batch_job_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The fully qualified ID for the batch job\n", + "batch_job_id = request.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split(\"/\")[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_get:migration,new" + }, + "source": [ + "### [projects.locations.batchPredictionJobs.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.batchPredictionJobs/get)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aSE_wqES_jAa" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_get:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].get_batch_prediction_job(\n", + " name=batch_job_id,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "LUy0NIF__jAa" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "a1CtiIM5_jAa" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_get:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/batchPredictionJobs/7156765165659095040\",\n", + " \"displayName\": \"flowers_20210226014942\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/6656478578927992832\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210226014942/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"modelParameters\": {\n", + " \"confidenceThreshold\": 0.5,\n", + " \"maxPredictions\": 2.0\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210226014942/batch_output/\"\n", + " }\n", + " },\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"completionStats\": {\n", + " \"incompleteCount\": \"-1\"\n", + " },\n", + " \"createTime\": \"2021-02-26T02:36:52.483588Z\",\n", + " \"updateTime\": \"2021-02-26T02:36:52.483588Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "def get_latest_predictions(gcs_out_dir):\n", + " \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n", + " folders = !gsutil ls $gcs_out_dir\n", + " latest = \"\"\n", + " for folder in folders:\n", + " subfolder = folder.split(\"/\")[-2]\n", + " if subfolder.startswith(\"prediction-\"):\n", + " if subfolder > latest:\n", + " latest = folder[:-1]\n", + " return latest\n", + "\n", + "\n", + "while True:\n", + " response = clients[\"job\"].get_batch_prediction_job(name=batch_job_id)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", response.state)\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " folder = get_latest_predictions(\n", + " response.output_config.gcs_destination.output_uri_prefix\n", + " )\n", + " ! gsutil ls $folder/prediction*.jsonl\n", + "\n", + " ! gsutil cat $folder/prediction*.jsonl\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_get:migration,new,wait,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210226014942/batch_output/prediction-flowers_20210226014942-2021-02-26T02:36:52.355258Z/predictions_00001.jsonl\n", + "{\"instance\":{\"content\":\"gs://migration-ucaip-trainingaip-20210226014942/10140303196_b88d3d6cec.jpg\",\"mimeType\":\"image/jpeg\"},\"prediction\":{\"ids\":[\"5133242657597816832\"],\"displayNames\":[\"daisy\"],\"confidences\":[0.9999988]}}\n", + "{\"instance\":{\"content\":\"gs://migration-ucaip-trainingaip-20210226014942/100080576_f52e8ee070_n.jpg\",\"mimeType\":\"image/jpeg\"},\"prediction\":{\"ids\":[\"5133242657597816832\"],\"displayNames\":[\"daisy\"],\"confidences\":[0.99999106]}}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_online_predictions:migration" + }, + "source": [ + "## Make online predictions\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_create:migration,new" + }, + "source": [ + "### [projects.locations.endpoints.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/create)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Ph5S0j4v_jAc" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_create:migration,new,request" + }, + "outputs": [], + "source": [ + "endpoint = {\"display_name\": \"flowers_\" + TIMESTAMP}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateEndpointRequest(\n", + " parent=PARENT,\n", + " endpoint=endpoint,\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cwUNXZaz_jAd" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"endpoint\": {\n", + " \"displayName\": \"flowers_20210226014942\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "yjsSo1cM_jAd" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"endpoint\"].create_endpoint(\n", + " parent=PARENT,\n", + " endpoint=endpoint,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ijvF_HGd_jAe" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "G4Gd_8oI_jAe" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_create:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/endpoints/3440574193450614784\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_deploymodel:migration,new" + }, + "source": [ + "### [projects.locations.endpoints.deployModel](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/deployModel)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "NFIRI0XT_jAf" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_deploymodel:migration,new,request" + }, + "outputs": [], + "source": [ + "deployed_model = {\n", + " \"model\": model_id,\n", + " \"display_name\": \"flowers_\" + TIMESTAMP,\n", + " \"automatic_resources\": {\"min_replica_count\": 1, \"max_replica_count\": 1},\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.DeployModelRequest(\n", + " endpoint=endpoint_id,\n", + " deployed_model=deployed_model,\n", + " traffic_split={\n", + " \"0\": 100,\n", + " },\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "g8FThQHT_jAg" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"endpoint\": \"projects/116273516712/locations/us-central1/endpoints/3440574193450614784\",\n", + " \"deployedModel\": {\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/6656478578927992832\",\n", + " \"displayName\": \"flowers_20210226014942\",\n", + " \"automaticResources\": {\n", + " \"minReplicaCount\": 1,\n", + " \"maxReplicaCount\": 1\n", + " }\n", + " },\n", + " \"trafficSplit\": {\n", + " \"0\": 100\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "c3_4BVyW_jAh" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_deploymodel:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint_id,\n", + " deployed_model=deployed_model,\n", + " traffic_split={\n", + " \"0\": 100,\n", + " },\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "7NmySa8R_jAh" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "v6qJ5zcN_jAi" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_deploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"deployedModel\": {\n", + " \"id\": \"5165312113245159424\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deployed_model_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The unique ID for the deployed model\n", + "deployed_model_id = result.deployed_model.id\n", + "\n", + "print(deployed_model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_predict:migration,new" + }, + "source": [ + "### [projects.locations.endpoints.predict](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/predict)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "941306da2e54" + }, + "source": [ + "#### Prepare file for online prediction" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:automl,icn,csv" + }, + "outputs": [], + "source": [ + "import base64\n", + "\n", + "import tensorflow as tf\n", + "\n", + "test_item = !gsutil cat $IMPORT_FILE | head -n1\n", + "\n", + "if len(str(test_item[0]).split(\",\")) == 3:\n", + " _, test_item, test_label = str(test_item[0]).split(\",\")\n", + "else:\n", + " test_item, test_label = str(test_item[0]).split(\",\")\n", + "\n", + "print(test_item, test_label)\n", + "\n", + "with tf.io.gfile.GFile(test_item, \"rb\") as f:\n", + " content = f.read()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://cloud-ml-data/img/flower_photos/daisy/100080576_f52e8ee070_n.jpg daisy\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "6fb84nKh_jAk" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_predict:migration,new,request,icn" + }, + "outputs": [], + "source": [ + "parameters_dict = {\n", + " \"confidenceThreshold\": 0.5,\n", + " \"maxPredictions\": 2,\n", + "}\n", + "parameters = json_format.ParseDict(parameters_dict, Value())\n", + "\n", + "# The format of each instance should conform to the deployed model's prediction input schema.\n", + "instances_list = [{\"content\": base64.b64encode(content).decode(\"utf-8\")}]\n", + "instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + "request = aip.PredictRequest(\n", + " endpoint=endpoint_id,\n", + " parameters=parameters,\n", + ")\n", + "request.instances.append(instances)\n", + "\n", + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0SDSqWD3UILw" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"endpoint\": \"projects/116273516712/locations/us-central1/endpoints/3440574193450614784\",\n", + " \"instances\": [\n", + " [\n", + " {\n", + " \"content\": \"/9j/4AAQSkZJRgABAQAAAQABAAD/2wBDAAMCAgMCAgMDAwMEAwMEBQgFBQQEBQoHBwYIDAoMDAsKCwsNDhIQDQ4RDgsLEBYQERMUFRUVDA8XGBYUGBIUFRT/2wBDAQMEBAUEBQkFBQkUDQsNFBQUFBQUFBQUFBQUFBQUFBQUFBQUFBQUFBQUFBQUFBQUFBQUFBQUFBQUFBQUFBQUFBT/wAARCAEHAUADAREAAhEBAxEB/8QAHQAAAgMBAQEBAQAAAAAAAAAABAUCAwYHAQAICf/EAEUQAAIBAwIEBAMFBQcCBAcBAAECAwAEEQUhBhIxQRMiUWFxgZEHFDJCoRUjUrHRJDNicoLB8EPhFlOS8QgXJVRjc4Oy/8QAGwEAAgMBAQEAAAAAAAAAAAAAAQIAAwQFBgf/xAA1EQACAgEDAwEGBQUAAgMBAAAAAQIRAwQhMRJBUWEFEyJxgZEyobHB8BRC0eHxFSMkM2IG/9oADAMBAAIRAxEAPwDRaXdr4ajO9KmGjU6Tpj32GLKqn1O5q6KXIpqIeGFeMECN/wDNvVgD6fQXiTAt4SPVY1/pRaAZvUNFtJHIntVBP54xyH9Nj9KravkKAH4ctWT9zKVONhKM/qP6UvSmEXXmi3FkrM0GYj/1EPMv1HT51U4NBGukypqdp4JP7+Md/wAy+tOnaoVizU7FoiTjcVUxgeHzoCPxJ1+FLZDWcLj7xE6L1XfHsa0Y3aAxxPbHrjcVcKBzW5I5sZx2pWNZZAAMDpjcUiAWNG0brKvzpiDVW8e3BxzD4VLALLleRyo2rPYSu0XxJghG/UUYche41ij8EHO61o4FKLqPDZ/Liq2yCyUb8h3zuDVTe4yJpIIkKsdiKiZBBqbYlfB2NJYQjTYC7L3CjJqJ0iGq0Fi2GwRmrcX4bAx1O/hxuParhTL61ckRMo/Edtqrk+wyMjOVTHcmqmEE5ERiW3PYCoiDKwhkunXblA6mrVuBj4IsaBF+dW8ERF35aAQbxjkcq5+dK2QKil/i79hS2AIWUDfHvTAJCZpNvy+1QhJVyCahKJPIIo8namqg0KL25MhO/WllwQT3kot4nOR+En9KqQTn2l6q3jJEiF3dgqL7naqISbdDtbHQdOnuLIZRhKF2fAx9K0x2ENfpmuEoGVtxuff4irrFNBaXyXZDKQHP5c7GnTAUahp0V4CCAjn2qEMzf6S9tk4OPUUGhkwGO5ltmPKxHxpSFMlpZ3kgmiJsbxTkSQjYn3X/AJ86VpMhKUffI+WYIk46vHuje47j4Gq2vJEI7i0a0mLAbelZ3sEusL+XTLpZoH5W7HGQR6EelSM6ZDe6Tq1trkWABDcD8UZ6fKtkJqQrR7LYmMkY2z0IpmAGMPI242qt7MJb4DBBg5Rv0qN0QusWMeV9DQTIUXyBpM/rVEmRFFsfAu0dh5Qd/h3pYzphNBNbeGSmNjW1+QAM6flPYbVTYBTPFhwR2ql3Y4LeOBCGAwRUsginzJIVPc7UCDy2QWlk7d2G1LN1sQ0GlJ4NuuAM4GK2RVRSFDbqUeHnPbBJqN2Qx2rXwMp5dz0x6VU2hjOyukbkndjSWQ+trfxZsk5LdBijHchordBbxqifiPX1rQtkKEdF6/OmbGBpZM5AOcUlkIIzIpJHwFK2Qks56tsTS2QvjDOc5yKYAWgwegApluQtB7k7U9UEXXV5zlsnyiiQAzzAtnIqlu2QyvE+p+F+5BHiSDdfRf8AvSN1sFGN0gG21W3c9mJHseU4qiCqVjvdGxt9Rkgl50Y4I7VqWxWPrLUecB4zyOOoHQ06INbTVsPhm8N+pU9D70yYKNJYa4sihJnyv8fdfjTpgoPkVZY9/OpGxHenAINT0fHM8S8y917ikaCZq5iMMmMEDtmkYwG106MHB36HFVsgUlxHex4Y4bpv+nyqmSIAzwGJiuCMdKzPZkPLa7ktZleN+V12B9qaM6DR0fh7WY+ILQBtrqPqp6/8/wCetboTU0I1QZPYc42GGB6YpmgFSQ8oKEH2zWdt8BPktmdgY0Zz/hUmjbZCi7tZl3e3lX4oaVpkBWQoQrKVJ7MCP51naaCaC0f73p8ZJzJH5G98dD9K1wl1RA+Qa5iDgNjfpQkQT3C8svSqepLkYU3+QNuhpE9yANtbmWcA9KeG+7IMdUlC24RdiMAVnlK2Q0tqOWJR6Dv7Cuk3tQoDf3/4kzjG1V9QaMhdyEuf4jVTYQF41E2W3NQhZdalFoWnSXssbSEERw26Y57iViAka57sxAz0G56A1ZEg/wBLt5kto/vEiSXJXMsiDCFjueX/AA9h3wBnerkAslDISCdvWq3LsErSDlyzZwenvUshU48pyd/Slvch7HADhm+Qo2QNXlAAG3vTrchaCScf8FWIgLe3YC8imiQUSytcSCFNyetBvYgJrusR6NbhQA87DCR5/U+1Ut0Q53d3UlzcSSyuXkc5Yn1qssSBpnMMwdeqtmk4dgNXYEXMa4xv/KtUdxBhFDJA+QfnTVRBlFKbgAN5ZR39aPJAmC8khPI4+DCpfkg/0nXQmI5D5cYz3p0wDeQiRA6tkflYURRPqlgJ1Yrs3f0NKxlZkb2Jrdjlcb1Uwi83wgkIZQ0Z2ZT0I7iqpMg0S4SVUiZ+ZJBmCYnr/gb3FUTQUUTRGNyCuG7iqG62YSzS9Vl0q9S4hJ5lO46ZHpTQm4uyM6rb6zaalpkN6kiqrjcZ7/Cuj1WrRXVAN1rVsoBy2R3ApGlyFC6TiWMNlVkI+IpHH1CTTiWJieZCp/xCq+l9mQKh1KK4XySMvwOR9KnVKPJAi31Uac5EsQMTjHiR7fp0zVkJJ8AaDZZElj8SNwysdiKktnuATXgzJmsjdsYVzLzBgw6UtkKraFYwSRnO1Wt1GiAWoz5uY07F1H6isqdyQTWGUquew/Wum2LwIdVu1jlB653qu0mQQXUhZy3rVblvYSNvbtPIpYbnpUjbdkFukuOL+LZbmI82jaA7W0LfluL7H71x6rEjCMHu7y/witUQG9RcrgCnfBEfTL5VBGHc8qYGfmfpWLLPp+bHSsjcwsG3YACrE9hQRkOQSduwp0Qn0IUYz39qKIWYVuRtvL0qxEK7y9EAIXqe4pyCa6ujgkneiuSFD3q6TYPdSbSyjyjO+Ow+dI2QweoXb3UryyHLt136VSxhacs1AI+k4UeckrdRYPqDVjxp9wWNdK0eXT40jkkSbGcMmenoc1dGNcCj+O18SPbcjqKeiHn3c5GxDetKyFqjnYo+xx1qEIFHgYc3Q9CKhBpY6o8AUEnB9DsaNgoYvOJ4soT6gelRsAi1GHx0PMMN61XLcNmV1C2eM4YZHrVMgg1jeiLmgmyYHPTup7MPf+dVvfkYdw3HigRysGcDyP8AxCs8lewAe5/ctzEEKfrSwi26YRlpOrr4YSNl8P0U+tb1tshQye5ZtwfgKjIDPcyL+Un5UtkIHUCrDm2B9RQIXRagFcNHJykehqEG9rxFj91cAMjbE/8Aaq3HwQPhu302YMjB7WTqOoplK9mQMu2DFXU5RtwfWsk04SIAzIGbaqk7YQaZxGygH41Zkn2IJZJBNqtuhO3iqf1FU43ckFmtvJCIio9a6E5UIjPagyk5PmYdqrvuwi6Zi2GK7noKDduwizia51CLTUsdJcJrOov92t5SMi3BBMk5HpGgZvdjGvVhm+C2AajhnQbThzRLPTLFCltaxiKMMcsQOpY92JJJPcknvVyIOkiJGKLIGxMrxEnAVRjmNZ5q0QVzlbiRmGeXtUWy3IDSMvMAD0/SmRCK8oztzfGrEiELu58GM8u21OQTSXLyNzfQUyIViFWzJM4WFBzOzbbVG6IZXV9XbVbpn3EKnCKfT1qhuxhNctn41AgksohUsetTjcBdHrU0bgh2BqWQcWHE7xMjMobHrTqbQKOg6PdW+qW6z25GD1Xup9DWpNPdCsZtYCXoMGpQLBXsGOx2I9aFEKHtmjUrImV7EUvAbKGhMQBVSVPalaoJdBd+EcEHHbbpUsFnl1Isw5h36ig2AR3oDZU7/EVVIYz11EY5CQNhVEgottLsgAHoPWq3uhi+/uWnZN8gDFNDgB7bXXggDBzVl0BjZLxmXyjBHrRsBW95MDny/CqnkDRR+0iPLJHzD/Car98kHpZAXEUuSmUb6VPeKXAektS9aMYcbdiKnvFwyUNdL1rwD4MxLW7+v5femUk9mBo0ljNy5gYgqfNGfejNdca7il3MCx9e9c+MviQRPcy5lc9hVM8jvYZKxJbTCTWoAeqyL/Ohhl/7AtbGtvrgiTlUeYiuhKdMShPIMM3N1PrS9RKBiyxq0jkKiAszscAAdTTRfUyBeiWryyNdyx8ksi4VWG6JkEKfQ5wSPUD+EVu9ADyJQuwplsQkkyLGR19TSuWxAaTUVk8o2RKTncgJJdg55f8A2qckB0lDscZx71YQquL7wRjqx/SjZAD7y858wO/anSIFxWhVS7+XG+D2p/w7shzXiT7Q7bU717GykBsIWw8wP9+49P8ACO3r16YrO59UqCtizT4LzV0BsrGeZT+ZUPL9Tt+tOoNjNocW3AWr3QzL93tB6PJzt9FyP1q1QQthK/Zdbtg3t/LN/ghUIPqcmm6Y90K7ZiL3T5rVzzDIHesrQ9lEU3I25+tJY1Gh0HXZtJullhYAnZkb8Lj0NPGbjwBo67w3rdrxBCDA4WZRl4WPmX+o9xW2M1IqaHbWCzDDdR3A3pqsADNpRQFSeZe1I0QVzabNCuVbnT09KFbDWBNE5cldj3BpKACysyMduXHakumMAX3nHMPxe9I+SCK6QlyDvVTYQIgwuCOlZ3sNyFRsJo+XIB7UVNUSj1PId+1JLJQaDIw4wWIVW39aze/6nXBKGptxHGDjIPeqMqnHuRC65AQny7ZzWJzl5LEgJmRm2PKwrNLUThvQ6iqPvFYcwY5FWx1qkluDpLYpiuO4PetcNSm6A47Gn0K/M0IiLeePdTmurHJ1Kyhqh2LjLu2NiucCubmn0ZGwpWJb6XDMQe1YpZLLEhXpI8XUw+c4cAVZjyfEvmRmnupT4pYbnNa/e3JtFdAU6s5329aZzdkoHTknuRalOdVUSvnp+Ly59d1J/wBNdLCtrFZpLOApsPiTVylbASnxFzHvjrRbpEFQuD5hkAVQ5NkBTMrcwzse1S3VEK/GypA6VbHZEIyXXKuFH6VZdkAvPLITy/WrEiDSxsgo8SQhVG+Wprogn4s0674uT9mQXjadpTY+8zQDmnuB/wCWudkX1Y5J2GMZy1J8gPeH/s90Th5V+6afEJF/60/72Q/NunyxT7LghqETlUDOQOg7CjYT4ycucn51LIDPcxhj5t6lkOQnWvFTlkHOPcVk6rCCyx290SUPI3pSvcK2BiZLfB3dfUVVuhxjp+sS28qSQyPDLGcq6MVYH2IoqdApHT+F/tYUKkOroXxsLqJcn/Uo/wBvpWqObyI4nR9Pv9P1mBZLS5gu4u5icNj49wa0KSfAlELnR8czQk79UPeiAS3+jliZIRyv3U0jW9kEV9a7jmUxN64qqSGQouYSr8rPufwse9VN06YRPc2+5UnDiq2QWyoCO21Z5+R0UxSeG++4zVHCG5GMYV8c35h5WFZ59URrK2nktJPDkHMn6/Gs0nezGSsbaXqkaoIZmzCfwyH8vt8KMM6/+vJx2Fce6J39u0Jwd17MO9Yst43THi7QjnwrkjbPasrlvQxTDclZQrbqTjFcbUq03B0y2Ndw4oFGV6H1rDpPaE5SePJtJDSiuUFaZdNb3CEZBz+lev0urt7szygauWcJE7Z2K7GqtfqVCN2JCO4o1ScRwsx7rmuWtRdblvSC8OfijfP4pM/r/wB60Y9QlKKsDVo1EAE0oQtjOSW64ArctRHHjeST2W5VV8AXEdxa8N2FxqWo3K2umW8LXEtw/wD041BLEjuQB077Cp7Oz5dZijlyKrv/AESaUXQu+zpbnUdAt9WvoTb3urY1CSA9YEkAMMH/APOLkUnu/iN1Y16nqpJIpNqgwc9hRjKtyCnU5mVHBbGaSU3RKFbTYTcnpkmkbukE8jMZHlz/AFq5cAZcqBxgDar4pMBNLBnBY4jT+JzgfrVySAVyJFGxEZ8dvVelWWiEo7KS4Ia4kwnYCpaCGosNuuExgdzUshXJeoucMDUsgLJqbk4UZ7bVCFDyzynfI+NG2QryAcO+/sKBDk7Wjr0G1ZByAidGPehdEJLMVyrrgUL2IRADNgAg0gxZHK8PQ5HehZBjZarJaSrNHK8Mq/nRiD9RTKTXAKs3GhfarqNkFS4f73EOpJHMP9j+nxrRHO1yK4m80vjzTNaUb8kvUgfiHxU7/TNaVkjIRoLubO01SNhHIr5GxU7j5UWrQDKaro7xZVxlR0NZ5QvkazO31t4RDE5Tpn0qqUd7IK7m3KjIIKN0IqiSoZCqRmhcBqyvbkfkY6dKt4phz5wMj3pdppxDuty4j70DbykLOn4G9faudk8PkdMXJcSWkhBBIBwymuXmyLhlqVjiw1VVUQTkvbHdG/NH/wBqz/1ka6MnAXB8or1KERnIOVYZVh0NY8mdR4ZK8iaf8WAwJB7VwX7QWbH7yq5/Iv6KdDEXahkR2GSMCvOR1cpyWUvlHp2C4PMy/GvQ4PaCW9lUoWaiV0itIjK6RqVxmQgLnqASdu1cv257SnkhCGJ773Q2HGk22LOJoGghgRkZjKmF5AcNjG249xXLxe1pzx1HmOzvkd4knuVWh/Z9xbWm3PHHzNjpktv+pP0rt6DW9dKLutimcaY+iufukKswzM+By/7V7TNkwwxxhn3t8ea3+3cxqLb2OA//ABF8XTcVcT8O/Z5bzswvrqB9T5G6Ru4WOI49uZz/APz967Win1w9527fJFWTZ9J+lrCKOCPCqFQbIoHQdvpXSjlvkRrwTnuXWE4HL8ad5tqRKM7ezGV8Fqr63J0NVFSoMAA+XuTWqKt7CE/GjhGdvnsK1/CuQckDqUr+W3Tc/mxtT9XgFHwtWkPPdTMx9CadOyFrahb26lVx8BTWQpOrSSbRqfQEU17EPOZpfxty/E0UQhI6qcZyPQVLIVteqBiNMH4UbIV+LJI3fPoKKRAiK2kf8u9MQxK6DfjYRK3t4i/1pHiYbPJdBvVUlrOb/SvMP0qqWKQbAZrQxNyupRx+WQcp+hqpwa7Bspa0HXBWq6phsi1ucE8mT6rQCQ8EEb7j261AnyqYTlWJHpU4IExanyEEkqQe3UfOjZKNPpXGNzblOaQzqPzZw4+ff5/WrI5mitxNzpnFlvqsAWdg22OfGCP8w/3rSpqXAtUVajo6lWaICSJhnHUfKkaBwZS+07wwwXIOd1NZpxGQkvbfA5geZcYPqKyZFQ6Yu55LSdWRirA5UjtXOnNwdotSsePIusWRuoDyXcWOdR6/0NZ8uVSXX9wpdge5db6BJQCLkDDKB+IDv8a81rtTjhOML+JmiEW962LGs3t0EkavJbsB+8xspPrXh5e0+uLjPaW+xu93vsGSWlzp2mpPcxCaxl35kP4M9Dntn6VxcWunC8UJfT/BbLGnuyNxcWOjrFaXCJiceKzgjKEgb9Oo274GKyp5crcot7fmPGG1gOk6bBJrVrZzP9+nvYlWyaN/zvktnHTHIcE7YJ7mr5Sk8bcFVc/JfzcpUPi3Gttplzp8VoLyURXsjlJbVxllIYjIx1GBn3HSqlqKk+jeK79gJWtzSz3Ea2DXdzarJYwRu4EqHJcEAMuQMgAvk/GufOeTLLp6vif6clypAem6p+2LMh/ENus3hpy7Dm5CRHzdF+dCeP3Uvp+/JoyxiqSIaJoUtvNPPfMsl+GCP4T80UYUHlVdtz5iSfUY7V6H2bqMcM6kvwR337t/8MOSG1vljOC1N3LO8XNFImYhNMvkHc8vr79egrZn9pT1GZzbSj5sSOOlSMpf/ZbFb8Z6PrNrfW8EEbPLqcMuZHvp85ilyy4UrsNsDCrgArk+19lahf07lLOpN/h2cUl3XLv67+pkyxqVKJvbbUf3ZBTkkU4KntXex6vp+GXJmcQPUL9pMID8cVpjn95KkDpoW8qHzE5rowqPIjKpLnmPKg5mHp0rbHK5bRFomLfIDTHm9B2FWxdAKrjVIbNSE8x9jVnUiULJNQnvmxzcoPp0FOmSqLooEXzNIPnVsWAIWYLkR7E96e2wFvhzPsz5+G1OgFkVlcE4IwKJA2DR1c+YNn1A2okGtppCIM8mfciiQOW1ghHmKgntTWQwT2Eo7fKtFAspaF09vjShKpJ5QhRjzL/C+GX6Hake5AF7OznODGbZvWLp/wCk7fyqqUIvsQEm0OWEGSMC5iAyXi/Eo916/MZFUPE1wNYCYFb8Khs1RQbK2tzjBAUds0o1g7xcmcrk+lI9uAlTuynmj2I7Gl6iBNpq8kEgfdHHcVFJ9iVZq9J4wZAAZOX17qflWiOYr6aND+0bPVVCzKI3P4XXoaZzUwUJtW0vlDIcYPRxVE0mqYVsY7UYzCSrNhh6964Oo2VGiJ9pTzW5a5hbmCEI6LuSD/ttXkdXr1gyrFLbqT37fI1wx2rQ/QWkhs7qOeOINMhMc2Qr+bcEjpt+ux714LW6zPkzShL+26rtsboQSj8xjxM+i291KZdaOitIuAskRkhB6csgG6g+wNcTTrLlSSh1fJ7/AE8/Wi3ZMqa01yHS7a/sLURTWXKlxYZxFLynzlX6crAq3pu2+dqsrA8jhkez4fj59/P5Eck+COo8J2qcSRcRc8N9p+OWa0mXm8KX8KyL1DDK53wckdc0cWpawPDG1Nd/T9vBVv1WafjfTI9e4VklvbeaxabFtNdxQq0xRjsCPZ+RhnGCB61j0mXJhzR6N0t0r79/ytP0ZG0twC24ebX9c0u91G7K3NgWiuROpBuUYADnTfkPU7ZB9qeeVYcc4QjtPj/87/mv0K0+t34NJLYw3mn3Gj6u/hXsiPHJd25LchlQqSmcAJh8AYA/nWF5HDKp418KfD9H3/6W/iQw0y3lsIZtOt7eB7WGMxC3j8qsM8wIJ35s+o6n1rPOXvJdTe77/wA7Fjews0ThuK3gFlb3czM5Zk+8YJQ5HlYjc5bmGcbZ6bVoyalt9TVdnX88DbvkSW2u316bNGtpLaK7PJAsgIWU5GfoMn4AkZq+WGMFJXdc+g8acWwxNcsr+7aCGCSd4mKkKh9cZx26e1WRlnwRUrr6lNRldF1yxjUsoTB6MnpjYV3/AGZ7QzSyuWWbexmyY1WyEV3qQRygGT0wK+jaPVOe0DFKKRXC01wSWBRPTpXoMXU/imUMLDx20YONx+tdGOSIjQtvtSeU8q7D0FaVO9gUBqqk5Klj61agB0FuZtk2+O1aUkhA620YlslQx9qsS7ksZw6QuQPCwfXHSrluKGJp8MIzJLy0/BC1JLRPwnmxRRCZ1NEGI0C++KjIQN9PJ03HqKYh5Gjy7sAKKRBD95JwGdfYmtQp65Qjzx8w9RQCgO4ghYeU49jShFtzaFd1AI9qRqiFEV09s4O4we3alIFPa2Wsjmdvu1z1EyDZj/iXv8Rv8elVSjGXJBTf6VPprItyvkb8EgOUf4H/AG61RLHQ1gToiHp9apaoNg8kZB2A+IqlpoZMEljkB3x9Kr4GKiJYW5lIweuKVuhkFQatJACrESQtsVPQ1W50TpQxg4r1XS4Q9rB/4isV/vLCRwt5EvcwSHCyY/8ALk39H7UVnV7iOHgqXVuHuNef9kapFHfqOaXTL4G3uIj3BRsEYO3pXM1c8cV1dX3LMalYRpPDdxFZzlJkhvecbSseXk36Y75x618h9taxZNQotfCv1/wdfDGohg4eMdpLfXOppZq37uSFYieU8w/eK2eo6gAdQd68/wC+uShGN+vp4LlHuaXRbTTNc4gttW06VGks42hWS4j5EeUpjmKcuxCk7qceYHAIGMuSWXDCWKWyl4fC8X6vzv27gY34Zl1eW7uo71rW0un5kMEziRZCCehU9Dn1+VUzUMaqEthWxRY2djJp01u8F3b6jbsCieNvFKAD4JDAAqp233wAc75qyUmpXtT7+nnvz6CW+57fX+q8LpZQ3UsTWlySFkgk5iJcfhwRzZO4G1T3EcvxQ5/YXqVBc3Eh1+/tLSKVLeay2uri5DBQPyqcDJOdwPjVXuXCPVPh8E6kuC6GcW2tTDWUE935Y5GilYQGPbleMDBGc533HTJpJwUYpYvw+vN+GMpXwWQz6kdSvf2RaPd20TMYnacMxIA2GTlhkkb77UsseOUU5um+w6klyH6VcRajZ3MqmRb6eOSRJpCUKuQSF5egAJ7gnbfJqifwSSktk1tz/LHt9g7S7v8Aaui8lt4NxE8XLDOx745QebfO/eqpRcJ09n3C2kZtNRu34hk0dYUs7tCObkA3UDZzjsR0NbJY0sazN2v5sWJLpsB1PV7aW5XNyLiRmZPu9ru0hJwrA7AHuQNv51r0+LI5KONbtqr8lb4bfBfJCuixLFColkP97OfxM3f4D2r7Fhy6f2NijHa3y+7f+DkyUszILcyHZt8/xV6DS+1MGqVQkmZ5Y3HkFvYTJh4jsPxIe3uPat8sMfx4/sJfZlC20B/E6g981fjjXYDYfa20LDyOj+wrbFdyth8SW8OOdMH2FaEgWW/fFQYt0Oe5YdKsragFck1yNxJuewqxJohWplc+YFm9KKV8kL47YyHOcN6AU/SQuS1MZwyHPrRqgWXF3jOEQMKNEISSzyYUAp8KaiWKneGQKzoPfarwbkGiidD4eV+FQJQ0DLsHB9jSkBpY2XsCPallYRbcIjgnPm9KrZAAqyPlWwRVb2CMLHXpYFaKVUngbZ4pV5lYe4/5jtUUiUHw6HpevEfs+5+5XR6W1wxKsfRX6/I7/Gh0Rl8wAl1wrqFueSe3VCO5cA/0NJLG+5LFcuh6gAwNlJInqi8/8s1mljk+EOpCe4tDCSkytET+WRSP51llBrkdMW3EQibY5WsUti5ANzcm0XnWQow3DKcEVgyyGSLtBmtuL9ZCPaxNrAiMYvVQKZY+pRjjG2M/BTXi/bmTpwNzeyf+jbh52OgaVCwt7eG7vYrHUHzF4F6hVQ+4XDA75GMetfOsmR1KMU5R523/AOG2q3Feo6LdyaDBbaklxp0UXMTq3heIkoOwXGQWyehx0OPSrYTgsrnH4r/tumL1NLYusNN1TgNrbT01K2uoLlnNtf253Q8vMw5d89Njnv6UMs8Ws/8AZTTXKYrmNNQEvDll+0LG/juxJOIJRekA27MfxjHVfbt9azxhjz1Fqtm9u/8AOxT1tcg/EGn2VtpOn3OnLcS6oZ2hdWnb+0FsnnJY4TlHQbDAA6ircU3NuM6Sr7V/n9RG3yEniDVbC+tor2B7NPD57aOR0kjyThmBBODsNtuvSq3hi4/DK/uJ1J8BWpcT2fGGqW9nd3TWcdkeV5ebExdhtyZH4dubJ2bII23oRwzwx63v1cd1S/f07CSlXBTFa6facV3MWrSPqs0caLBOGMB8Jgckcj7NnPm6dMCpJtY08KqNu1zuvpwWRlaPL3X9N4A1uOxsXCWUipNHA0ru8btkESliSM45hk7jOMYpHgya3G8rW/F1XHiixNLZmha50ix0qKYN4l2h8bx+dgzuXyeYZ5XznoR3rB/7JycapfLj5eC1X5DrjV4IYoXs4eW/mZUWxg5VVl6c4XYDAAJxtt69aowlNvreyvd/zcs42GUdwRC8GrWyOt0hUgEFSCPwhhgjG2wxgmqWnFp43x/LoZVJGR0RNMGu3Vvp9pbRxwvym6Ul3kddsIWJIXO+BsSM75FdrHqMumljz5N6d1x96XJFDri4rgX/AHy84k4hvLbTgps7FuSW6ckRhvQHG7ZB2HoTsOvroaXUe2pqbfSklzwv9mFzWLZbjGexvUjHhtHc8pwVVsE/AGurpfYefSyeTHkUn44K55lPagNpiJhz5iY7MrjlIzX0LTZpdC69mYpLfYEksmGXclIxvzsCFHz6VqjK3sATHjLhq3uzajX7O6uh1t9Pc3sw+McAdh8xW+EcnNMrdWaG01ATrzQ2WrOm2HuNOns1Od9muVjz8s1ojGa5F2GEV0FRWYNETuVzkj5ir0muWKXR60sewy49DVyaRAmPiUDpBEPcimtIFF68RFx/cRk+1NaJRP8Ab5x/crn401olHo1sNuYBt6GmtEo9XWoj+KE/I0VQKMFFqUyYIY7daFjBkOsOMEJ8aa0QIOoLKvNjHtmhZD77zFJtzHmoOmQBukbJIG3qN6QgnvIZMEqWPwqpoKFzySxebdj6Gqm6CTt9VfoVx60jkw0a/Q/tFmslW3vB98t+nLIcOPgT1+B+op45q2kBofm003ipefRtTFjef/bzJkE+65B+an5U7Ucm8WLujGcS6hxTwlDI+o6BdanZp1uNH/tkePdDiRfgUrHNZo8bj7Mw1l9rfAfEEjQSyRWdwv4knVrZ1PoewNYckn/dAZV2ZdNY8NX00c9tqU0sYYOYlKTK4B3GQQcbYrzHtDPGGOSimn58M1402zQaNrCWNvLJw5pskaRNzTmGMsCN9mA36ZwQNq+SarFkyTrUzv6/odWLSWyPdTsJeJtQtBdSz3GkXCuxFpKshBA2GeqgtgEkKRnt2pxuOmUulLqX0/nkeUr2NRY2s+s6A6X9+9iDGQIfD57mEpkLzMTgbqDtnIPUdscpRxZaSv8AR/z8ipt1se6bqeiW/Cstnq9layWUMXMHePLTPj8Q7q3NuMZ60rWZ5+rDy+3+e3BXS7mcbUNAn+z++S7hsnbwh92uLt28UyAZUrJuVbIH0rfD+ohql0Xu90kuCqrC10DTr7hOa6jW7m4htURo57ifLGQ4HKVDKgBYlScYHXPco88o5vd7LG347eb3fr+RTK63Yk4w4b12Ph2zv9aXT20PTv3sjQ3niTKGAByvKAQM4OCeu2d8a9NPD1uOGT65bJU0tt+Spt0J9EsTqWiXOvaXqEVnGIzcpY8gYSRqMdRvzHHw23NX5pKOT3GaNvi/Vips01nfaDrGmxWDaYbi+lflsyJQLkSNuGMp7Z9dgNum1ZHHLCbktl38JccButynhTjuxvtL1HRNdERkmbwNQ8aX94zrsrcw6gdiAKXPpsmKccuDdf2+CxSvcafZnZ6Lr1gdKvIZbm7Tmt5JGlfx4ix5jIA34SNsHBO/yrPrsmXBk95HaL37U+1beTTFmo4Rs9J1DS0WWUT3RQlb4yczREEgMMbAYPQep6Vzs0545uLVR8F6b5CLDU7e91VdMmu45I7CRjLjIWaQbLy+wHMT8V9N6Jwljh7xL8XHoh0k+Aq/0/QNLtbG5sbeXT42mEf9nHMckHH4m6HGQfb3qYp5c0pLLK9vH/Bm+lJFdx91ttFmttD06O2XxEZkgQLyM5GXbrg4GT17Vrx6rULKnkyuldWI4KtkLtOmtptSawm++REEj73cJyoWx2BOeU9M4Fe20eujJLFHJ0t/3N2r9fn+RjnjfNHuvQWd9YXGnz+I7OChmhlMbIvfldTkE7jbHU1732Vrve4F71XJP6fNeUYpwp7GYg4T0CzYMukWszj89zH47fWTmNekhqH2/wAFXSN01W4tYRBbs0EA/wCjCfDQf6VwP0q+OSUuRaKTJM5Lcu56kDB+taU3W4KKmadxsCvrirLAR5JHJGCfjVqbfBCQjlQ7ZFP8wFiLN/EfrTIBes1wuAelMmyFn3qUfPfamshE3kp7H5UbIKI4gp67e9W0QISFo2/FhD6bih6ELFteV8HofXal5IfNauhHL09aBD3wnZcFi3pyigyFPgyZxy49c96rtkBpdMLAt4e3U4pWiAE2kRyZKkqaqlFcoZMFfTpHHKTj0JFUNNj2Dm1uLVgQ5HoQapdxYypmh0z7Q9Z0hQryftCIbeHcjzAezjf5HNT+olH1B0XwF3l3wLx9JE+vaLp6XyHyPqFuhIOO0uP0JA9qj1GOa5piuDQn1PgZ9L1ZXsrHRp9PKHw420yNJVH+GRCMj4g/GvCf/wBBrVpIJSp35SZv08HPdE9LI1hbvT7VYdEeFAz8uIzMpJ8y7diPXv3r5RnfS/fTfVb48eh1tqpIOgurm7vrfTIdSgLwQjxLww+G3hggDKZILZYbqcHrgZ3rlGCi8yi6b4539H/nj1Ee2wv4htJ9LvY9OXWFkXU1kYXksPIyYfdTk8u++56e9XY3HKveuP4e1/z/AGVN0gDhe5suDNTeeZXgnAaa3bUMny46xl9sEknK+vvV+oWbNFVxtxX51+5nbt8iX7RJ9M1fia1nezbT9HltzK0q2zxxvcMSV5ywwSVBI9cnritukjlhhk1vNPzul6b38/BW32FMWua7qRWz4eSHVoLdkeaSSYI5BboSe4wd9+mat/p9PH49S3G+NiuUqVB/EX2gXNjeDhTVdJk+5TMgvVu35ofD2bOxyT5RsBnf0qvT6FdH9XjyfErqub4+RW99im94G1vXtQWbgl4TpTkNqFpPMEAcgsyR48wDKFGPU1dh1WnjGtYn1r8LX6v5MTdbIhpPCmqafHqGt8PWcz6MqhT52a5TB5WCxg83kJLeo5Se1TJmhm6cWZ/Gvt9+Nwd7RpeIePNFuOCooGjtop1mhb71HEGHIr5JdsZXYMdz1rnYNLqPfNb1vt60Wqh5rzy6tDYcQ6PY3s17beGbvVLFcNJbkAEvjHPt0YAlcZyAKx4bj1YMzSXZPz6eN+3cui99h3aWej6vBBDp+dKuyuBJp7hYzHt5nTo2wIB6+Yb7ZrBLJkg7yrqXrz9P5RfG13Hj8OCLULZdSlElvbKrWgtXxE/8Rf8AMG3AxttjrgYxvNUWsa3fN8/T+eTRjVlvG1to0OlW73EVzewSyhIYIJ+WaKUc2SrY8wx/EGG9TSyy9b6KXm1ar+eC5pt8WUaAsNzb3QGnPpbQx4bxJmlkCnGWYZ5cbDfahm6lJfF1X4VfQa2+diN5dm50OBy8c8olkiWaIYUqJGwy9duUqa9PDRwlgwSwx3fV1Xe7Vc/QxSy05WUw2Eb/AI8gnvX0X2WpxxpZElXjj+ehgyNN7BA0dSdiCPTG9etxpcmayQ0qBvxYWuhD0Fsl+yY1xyDmHqK1RutxbJLo4b0zVyaSAS/YUZ3OM0U/ACptHKn8OQe4q1epCH7IJ6bUbIQOnY6g596ayFUlhyYwNqNkKmsc7inIKIboMF54lf1GKtsgwgNrKuCjRj19KOxA+OwhlAKTAEdjQpACF0lnA5TGx/nQcSHg0eVTnwyPhSUSz5tJDrupV/cUrjZLK/2Lg4IJ9qFUErOggnKx8v0pXSID3GhPIpUxjPriqZJsgun4fbBBAPxrPLGhuoTXelMhwEJ+AzWOUH2LVLyIdS0e4KEraTyA7f3LnP0Fc3LiyK6i/sWRkvItsLvUNHiaK3NxAuSfAdGMXyGPL8sV4b2pjyyb64KUV52aN+JxS2ZpdC1m0vNHNvqZOmaldEqr3FrzRKchU8y5PKe5bHLzHtvXg8+BrJeF9UV4e/rzt9nubYzdW0ZyWxh4G117m5v0uNZ3gTTNMjEyOrEbF25TnIU4UbY6npW6Mlq8Pu4R+F79T2r6b/mScjU6ld6foph1K4v5b/UIp+cSzKfAiJXlOIztgHG+577VzsallvFjjSf3f1M8rCOJeO7HUb/Q9OmmW7e3vEu2mQeIIsAnDMM8vmwf1pMGmywhOdVaa32vyVSoQ/aP9rp4m4Wu+GNPtbnW7qceEZbYmVYgPzNjPLgZO+BtXS9m+zJYs0dTkfQo70+/ov8AW5VkltRh+FeHJOG7iC34dkk1KW9VXuFzylSCfMWxhVXJzjr2ya6mqzrVR6tRUVH+V6tmY6NrvAVk2soeKwGS8hAtLrS5giFVUcxIO7MGYDcYwdu9cXDrJY8aWnWye9rz+2waMBwzqmo8Ja3rFjoBm1Sz0+RjDeTzCNmOAzB2G53/AIc4IwM12dThx6mOPJm+GUkrS8fsyvgo4H+0viA2ck0P9svbyRlW2DOz+djhMjclTt6kGprNBgU+hbJLnb7/AFAmdV4as+EY+GdQs7iS6WCRC00s1w48IkEsG5R65O4/Lg153Nl1PvVJJX22HSFeh8UcQ8KcKwalPa3Udgk/gR3CqP3MXMRG5jzkLhQdwBuM4yKtzaTDqM8oRkrq3vy+6v8AwXqnSNQNK0mDRRrHDwTTbm0Rp57eLEaXyHzEuWI843we+QpwMYwPNlyy/p9T8Vuk/H27f9NK2G+hxxcc6bZalPqM1haqv9nt0ZSec4877nI8o8u2Bn1rFl/+JN4lHqfd/sv3Zpj5L7PQYdadl1W6gvo42kWFouaJkIJ8ysCQQwAOCp6bUjz+5X/qjXF9/wCV8y6MpRdpizVLyz0KabSbuMXty0eTcyyA5U7cgCj07nr1xVsIyypZoOl4/caXxfiYxMMdzptryiWJA++xYKSv5u4G3U+tdv2fqvdx93mklFPbzvz9NjDmx3+EsW0uo1zGY5k/iRh/I19O0WJ5IqeNpp+Gc6WzpkGluUbDhl+VelxxceUUNonHzOwJyproRTW5W2MbYYIUk1qjvyBjAKgwrE5PQ1ckAuS0dugDrV/BC9bEpuVwPrUqyHjacrZYZY+lMlSIefs4uDmPA9xRohS+nrnHhDHuKaiFLaODuFx7UyIc/wBNSC9H9nnjdv4c4YfLrVqp8EHNvprAklQfhT0Cw1LDbZSKFELZBLAn7pdyPxYqUAEGq3lq+S3MOvmGanAdmOdP1+1usLcN92c/mIyv9R+tC0Chz91EsPipyzRfxxYYfpQa2B3E93relW7sj3kAkX8SKwLD4gVnlKMfxMPIouOM9NjyUlV/gf6A1RLNBD0xXPxraNkoik//AKif51meoiuEHpYuuuPJAD4bOv8AlULWaWsfA6xiW84pubrOZZBnsXNcvUaick9y6MUmBjV2sibm4j++Q45fDDZHx6V8s9o6yWqXuotxfqdWEOjdgN7JBxBcQ3c8l3/4eRh46KpQzMDlo8jGAehYcxGdh0xxMd6ZNJJzfH+fpzW1l6d8EbfiDROLNVj0iztr+01GI82nXMMgt/ACKTylvMMDGc8u+wwO9scOfTQeeTjJP8V73f2+wjl2Prq1vtA123TVZ7TULUxeLEMcqyuSObbYeUZHTqdsYpE8c8beJNP9F/PUzzbYr4zvdB4OuvvtjCtt99BNzZQsW852VlXqVwSNiem9a9NDNrY9Ev7eG/0spe259wl9tek8L6bLp0cDxmXLixitmVmZhgkEAZz/ALinzeytRqJdbla4tu6KHRm7TVotK09dajQWuoCV3kjidihV32gKnGAucZ67Z7VunieabwV8Pl+i5vy+wqDNe+1e91LQRYRZgskEc819ATKscmTyhDjyg5weuzYyO9OD2bHHk9493xT228v1/m4/T3MRY3/EbtcDT5DZ+JgtPFugUjBPN/EQDsPbrXXcNLFqWTevv9vAkkqs6Xb6loXDh0OfQI4bYxo1vcPEqvNK22GYgk8xI3wBvn5efyLPqFNZ78rsv54KkqNAeb7TbxY4b6LQUj5ke6iAV7mTl8ivGR+Hm/E2x6YO5rJGUdHvlj1X9ku7vz47DfIqXiHUxpmq8KajbyW2rXFuYpQm9rhgVEivuGXbO2/brmllgxQnDVYpXBO1527Ndn2LosZ/Z015oesSWOqSRaxcadKj27nZArL5XUdCdzuc4I61n13TOKyYV0qS3+fg0Rdo10cel6jxLdT2UxtLZRGLiCzOFklyS43yAegPLjqe9cybyY8cVlVt3u+a7f6s0Qtohr9yeCpIZ9MjuNT0x2Eg5pUR7UnYIWKM3KdwGxntnO5mGMNWmpun+vr2X0/YtuXCRXwxctqCajeHQrJ7pgZgWUuzMcADnkfB7DYipn+Bxh71qPH8pFig0m5IbwXa6hpc9tNBPpiM5WW3KRiRcEEgE8wQEnr8cYpMeOSzLoak/Nut/tYkvw77CRNP1GC4kk++2tshclVSVmOMnHRRX0v2V7Olg6ZxzU/Bz8uTq2cRva3mpRbferS6Ho+Qf5CvoWnnmXMkznyS8Da3nlutpdNJP8VuwYfpXYi5P8UfsVbBKfdmYYeSJ/4ZEIrQukAxitS6jBD46Yq3ptbAsJjR4tgPpTpUQLSVwcMNu2RRIXpIuPwgGiQtBEmAwAHtTIJaIk9AaNAIG1Rj5t/lRIfnay4alZlMsywEbjlyzD6Y/nQWJrex2zZ6KPuahXurm5I6eIcD+Wf1rQlSENFb3S9eVQPfc0yAXyTiRfwgGiyCi/VSCCuD86RoggumdCeTPyqljEbPWL3T5fEt7mW3cfmjYqaCcl3ING4vj1EcuraVp+rED+8uLdfE/wDUBn6VHJP8SJVFbWXBupk+Jb6hpbf/AIZ/FQf+sFv1rPLDhn2oNsHueAtAuV/sHFiW7dlvISv67Vnlo8f9sqG62Lbj7K72ZT911jSr5uxjn6/TNZJaFviSHWT0EN/9mXE1pubaGUdf3dwu/wBcVz8uhyrfkf3iFC6JrFtdKtzatbKDzc7FWUEeuDv0rxXtrHPDhcq9H5+huwyUnQfDqkGsPdaXrD/dNPSDmTwogrP5ubl5s4Xffm5T8a+duDxpZce8r8/z7HQvsJJrqLQ9c0u30/TNF0wzFYW1CdwrvG2M8zhVG+B0G5x1rowvUY5rJKTrelxa8LcRxrc0HEPDk5uLPUdaggureGXMdrGS4t+bHKzk7EH07HFc7BmioShgbTa39fkVSRm/tL4YsdV1azudBtObV7CNWurG1j8rI2QCOwb0UehJ9a6mh1Mo43HM6jLht91+3qZpqkJuC9RvNA4ivJJNOnhvlaMM3g8zqwDEDK7g5P6j2rTqY+8xx6ZWt+5noV8fNp179pGm2q6aujBP7TNP4XhSBCMMnmxuf8WQNq16N5IaLJPq6rdVdr5/8LElaK7eytNFuUuJ0uL6Y3T8rLHIIbnm5hzKMBS2SNzj8LbUZ5JZU4xaWy8Wq89x0gW5/alvf6jPo1rNHeTPztalQ8IjP4SctjopGBuDmmTxTUVna6V37+vAGmhBw/d2eg61a3l1NAbe5mdpJYlLNG7ZJAQZ6MQNum9bc8MmoxyxxTTS28bev8srlGjqNjey8Q8QtcwXUmnbKXMsBka5xsCAT5QAP9sV5jKlhxVNdX14K6GXGfDUWi8a8PX2ja/cXTalItvdWuoOZDCqgEOoOy4GcL3OMb5oabPHNpsmPLjroVpra77ffuWxia7iTg/hubQ21OANHqmlckj3sszDnjDrzhs9chiQPXGK5eDVZ79y1tO9q71sXqNMb3mraVdaXZ2y+BYajE6x2s8IVnSM7kHb8HXr3IxvWGMMibbVrdv+eS+KdjfXbLTRwTqckT3Md5JGkcty0vnlUOuUBbyqu3sBv61nwTk9RBSW2+3jZ/mXtbOhbpWr2mg391aQwSRWljZpcxXAR5ZLkMvMSMDzgHYFRgZ36Vdl0+TMot/ik6rivvxt55H6pJfFwK9W4q168019RtNChurC5RhJPCyzle3PhchsDG4JxjcDFdLSez8c5+7631x7cf7+hnlka3rYx1nrKxpEI3aRT17gV7fQ48qk1kXBnlVbDmDVlbbf3r1eJNGV7DK11Yr5o5XjI9GIrq45yTKmkzRWHGmo24VWuRcoPyzjm/XrXXx6iSW+5S4odW3G8RA8Wxj+MdbYZ4vlFfSMU40snC5ikB9T/Wr1kgwNMKXiy1ZQBGxFOpRBRNeJLcj+6P1o3ElEv2/A35XA+tFUEsTXIDt4rKfQg0aXkhfHqiS/hmVj6E1KIctbCAn/AGqwJFbsA9CT8alkDrO9fmUqCPfriimAf21xLPFnlJPvtT2AjPaSSKSFI9iaVkQrudPLdTg+1I0g2L59Mbcq2fbFI0yFB091GNs+1VSQSlrN98EZqvchU9synfO/pSNXyFAV1asckJn4iss4+B06E11cXdrnwppIv8jkfyrnZHOPcsVMS3vEOqnI++TNjbDNnP1rzuqlKf4jRClwXwarp11bw3ElpHPeWxR3gushJWXBOQp3Uke3yr5tqMWfFlfU/hd1XZdvqdGEoteoVq+v2l/p0OonQ/JIRJI0sARi2P8Aou+Ry8w9T8s4rHjwzxZPdrJS7U7+6W/6Fn4iMXG13rnDF9cWOleAizpaSXF/h4zI56LynJIB5iMbZG+4p5aOGDJFSyXab22dL7/K7EltyJNKY8MazZraRz3st755JWkBkyoOebJAC+gz3rRkf9Rjbk0lHj6/v5MbV7s2nCXDknEet3et3Wqz6XfXcrNLYQLH4ZK8yq6u2cHoT8fasGfURxwWHp6kly7XzW35Arey/hr9nXp1bTuJLaJ9QUyLeXNzyF85yMuM5QjlI6A7b1Xmnlx9M9PJ9Lqkrr7efIK3pmGitdV0K3fhvVNNl0rhyS4K22qzckkZXm5kjXDEq5JwHIGOmMkA9lvHnS1eGalOt4q0/Dfql3S/4622M7e3A4V4mkjs9QW+vJWUfe5hkSRNupbpgpgjAIzgYzmtqX9Ri6pRqK7eq8fP5bB4MrxaNL0a7s4mVr/U95ZJSoCyBiSfOemSc4IzjAroab3uaMpJ9MeK8V6f7A6Tofy8VW+r8DSy2sV1DepGV8dgE8I7hsPnOPTGMn0rCtNLDq4xm01fz/Kinp7o0B+0HRtF4Jja8tp9U8SFVzJbkvLIoJXmZmGAG752HrWNaDNn1TjGSjv54Xov+DNUrHOh6PxJxL9nMlx9ynuljti5gurkPJID1ZRgnOCSM429zWHNkwYNZ09aW/KWy+Zck6tnUPs/17TtK0KzMtrbwosfLItwgDAHZgT3z6d8V5vVY8k8sknZrpciW21+A8bH9kzW2p6TNCI5NOuoTNCAwIkiD4zgrjPUbkHNbJY5Q0696umad2nTtPZ1+3gsilLdMdaPwtZfZxY26SXnjaYJfFguvu7m7hwN42CA5UdeYBfj2OfNqZ+0ZPpjUu6tdL+V9/Tf9wytbNmfm4ls7jiDUJ9Cjl02OaRXaaRgpklA3k5AQVz5Tg9Tk7ZxXRxQzYVjlN9Uo+O3pfeuL/xZVUXaAF13RtdvHh1uzOiawGKG+tv7i4b+JWICtn0YI3x619Y0U1qccZyV2vlL6r9aMM8coq4PYPn4VurdRJbvHfxYyHhznHqVO/0z8a7cNM2rg7Mbl5F5QhjhSjg4IIwR8avjFrZguwhDKoDA59q0xjYgSl04GTWhLwCwqG+OMZxV8RQ2C+5T1x7g1cpMFB0eoY35jVsZAoKS9PUE/WrLATF3nrt8KeyEknzupopkCE0u2wPJH82zWihTwaXbRk48PHp6UA2eeDFG3MkqRn3ocACYbxY/+smfansBd98V1yGz8agQZ2LEtyb/AM6UAOYGJJAHqd96VpkKza8/sT61WNZ7+y5JNkGB60riSyQ0OZhuhPypGmSyt+GpXBGCR6VS4sliXUOEiRgKR/vWLJi69qHUqM7f8JMjcioeb3FcfPo29kXRmZq70eYySQ6dyNdoCxlY7RjODjY5Oa+be082HHP3V35/wdbFC92T1ziq4ttKtNOvNKGpm3jXzJM0aSnuGVBzZ+DAV5zDgxyyvLCXTfonXyv90zTbWxTwzYw6pod1LqNnFZQzP4n7OiDQw2/L+EqCd2IG7HJNXarLKOVe7k3S52bflbLj04Kqsaf+ANPueFLfVdL1Sca1DGbm5M7GZZoCCwQLsAAOXBG+x6npR/WTWV4ckF08bbU/NlbhasE4Wj1zT9Curi2uLOS6ZXuIlaPbBAwoJIz0zv79qszxwTyKMlSWz35KxBpOqaTbaDeXHE5NnqNxzNLcXcbR805YvsADzKR0HYAbbVuy4suTMo6TdLhJ8Ljfx6lfHI64h+02HU+ArVbiynu9PuJVe7kngYpbx55iCM4JLqvTpgms2m9nTx6iThKpJVGny3/q/wAhr23EWstw1xFJZR3StoelmyZI7uMmASLzEMcMMEIRjJ3zj033Yv6rArS6pXxz/L9AWmI7949X4amlWGyu9Isp1zqVqIi12oflHhgAZJPLzYIHUDtWyEJYszjbU5Lh38Nrv8uy3BdrYK13hxtQFhY2Og3apOGNw8Bwjxr+EnBx1JwT3HSs2HN7pSyzyK1VX68/p2C93VF/CfAEPEuqalpuvffrWxsbdWSN/wB3IsrcwVjjZscu475+IpNRrXpYRy4KcpP5ql+gyXVydR1biT/5axw2r3SXEzxrFbmIcjEHylnjX8PLux3HTA3rzmPTrXOU47Ll/wCE/wAv1LUqqxvpnFml2elavNBIl8YYEZ3RRKI+UMAA/wCHrgY656+2SWmzSnCEl02/8F7aSM9wNxBo+q62t83DUVtdxv8Au4rCR18PbAQodmHUZPzzitesw5ceN41lbT80/wA+w2Poa6kPtTS4vuPoG1C9ltJraPFsLckBrdwThT1L84CkjpnpjFZsdQ0lY1cXz6SXn0rjz9x2oyd9/wBhLx3Z6po5/atvIl/Zqi/eLjw18WMDIDTKBynqAZFC55V5gDkt6T2TiWrh8Hwtcrn1tej8b1v2qseR9D33EFjxGl9+5vLa3LnbmQFOb5ZIr3mnwe74ZjcrH+mXv7KAFq7RRZz4LjK/LG4+VdzE5RdplElZp7XU7HWgsdzEskh2GTyv/pYfy/SuzjyxltJFDVcEJOGV5ibeXxlP5JMK39DWj3Ke8QWwGfTWtpCDGwx1DDpR6ekFg72p/hIHqKdJgs9+7PGMg5FWqPcNliNKu+PpT0Cwy3uyoHmOe4pgBaXGetMiE/G9wPjTJkM9HcXK7czgH3q22QtEs4Pm59+/epuQKgDORnn9siile5BraQTMQMF/Y06IN7fTn2JLY9DRSBYfHpefU+xOaLVC8l8WmBPy49sUPQgZDpkRxzLj41OlMDYcmlRpuRQaoDLkt41Jwv6UKrkh88CsDnAHt3qtxGM1xPxJo/DaE3lwFmI8sEY55X+CjcfE4HvWbJOGPeTGSb4OLcacbX2uJJHZx/syzOQVVsyyfFh0+C/U1wNVqHkXTHZGuEEuTntrbTOwhAMkS4yucKFBzynHQbdK8VrtNBp5OH+/k2xbew30try+13w2MrRtbuwlTzLEM4zylh32xuf1NeMyrDjxSryvqa921Yp0/QozxZI+us8gZOWGWYlowx7sT+EAbdAB3xWiWdvTdODnx/j6hlzwaHXrjU9CSK0tNRAhk8r2plTxFTOWERY+Q8ud6yYFjyfFlh9af51zuK9uC+eWxteHo5NFvDf2wnRhZQFJ2l3BKMobIOBuCQOu9KoZFl6dRGn5ey+ZVzuiHH/GmlapwzquhIlxqOr3Ub3cFrPG8DwxqCAy84/CkaZ5F3IViBjJrZ7P0maObHmdKEaTqnf27tvl8cPcSVIy/wBkVlol1pDW3Fl6l09nKWFjNMqRpGQFRyCBzYJJGGIyBtsK2e0smXHkWTRw2kuat33Xp27fuIk+GzR/Z3faVrus6jo2tpb8R6bZuF0trqFcXLDAd2DEjnwAAV/EOYjsK5+vjl0+KOXTtwk/xJPhdkmu3nxsh4rq5G3F/wBmz8WcewXXDmrW+maZiKbVLWB0BikXbmCDqTjcHBx2PfPg9oLBpP8A5eNyk7UW73Xq/wBwdDb2FhbibgHjJYrvTY761u4nEElhcKWkjV1zJyHfbI2PLktsTR6dJrNM5xyU0+6709vr9eB6d8EONuM9WsuIvvdloiLqGoxRyrdMjtOygsphVQSAylMEbkHfuKv0ujxZcXRlyWk3w9l3vz37hVp7IcfalJw3Zarpcs0mqa/bShmtdUg5I43HMP3ZVwuCoXOMbht+uRi0EdRLHOK6YNcrl/O03z8/8DtqMovcZ8Tz32l8N6FxLwZzPpYm/wDqeQoadVKgK4bzcpYujL0HkyOjFNF7tZcuDV1118Pond128O+efUbK22nHgaW+p6dfy6ld6ZamyaG4WSyu0JieVymY1dSPIFZuRyAQeUkZBFYZ43jcFOVpr4l4Xen39B07Xwlum/aBbX0P7O4j0w2pyFlCDxIg67cxXBZT/iXm69a9JoPZmhnPq3TrZ3tv/PFGeeSaiRTReJtPvY9U4f1SPizTQAoiLol7GMYIDEhJgd8qxR+uA/Svfab2Rhxt5dN8Mn67f6/QzyyrIumboWT8IadrcskmkxfsnU180mlXCGFSfRAwBj/ysAvoRtXWjg69mumRk6+n5C3w5NPka2u4ZLe4TYpKuD/z36U3u3B1JDdSfBfDPE4I7+/Sro7AY9sOIZrTCykzx/4j5h8D/WtuPI0VONmo03V7bUkwp8UL1Rxh1/7fpXShkjMraoNfRo3UvCAUI6YqzprgUAl0QJnyYJ9KPSSwOXTmiJBAxQ3CDSWZTcDFMQgMr5vT0o1ZCSzEoDg475G4qUQR22uEjBA+IrQmGg6LVwOqjFNaAHwaquxCg/KjZBpa6yAcAqSe2KOwKGkOroMeIT8cUbBQyg1IOVKjnX/LTdQKDkvnkI8qL8Wz/KoBhEdyW/MD/lFLugBiSliB/OgEB1fizTdEBWeUNMBtDH5n+nb54qmWSMOQ0c/4i+0DVtT54bJBplsR+IHmmYfHovyHzrBl1EmqjsWKPkwF1AmZJJCXYnzMxJJPua5sl1clq2JWvBd9q6Gdo2tbUDJdsKSPidh8T9DVf9LOW7G94lwUXul6bpkXLJKZlXcW9tlY/iznzOfpXI1WkwTVZF1enYshOSexmL3iK4sGK6VElmZB4eI1yzA9q81qNFp5r44qlubITlewqurzW0tIopuWV0yxidPNKTnBkPVjuMDIAAG3Unzco6aWRrEqXn/Hj1NK2XxDrWuDreLSzpEDwlhLm91fHiSkHHOVXIyOY4JBOF6d6x4tXeV5pXx8MeF6fzyRx6kNeDuF/wBsSwRaHpfKmiR+BFe21y6SSzOTzShMEMcK2c8wOQNh0p1WreNOeol/9j/C0mkl2/Tw+4IxV0uwh0rhG903jHVNSjuItW4m0+Ms13cyLJZ287IQxdznneJDzEZwp5Qc8vKepPVXix45rphPstpNJ+NqUnt5a8XYk1cupCX7Hn4ce9vb0Leajf2dlKn3e/gVoLlVKtlMDI2APKwyAM79K0e1Y51GMHUYtptpu1237eeNv1K1GnT5CIdF1G6u9Pi4k1bTFuJNYeV5YZcsLQkl0KqMqC6sY12YBn2AAwMubAoylpoOlDbbvwvnt+J7rjdsPS2+TosWi61H9rcItrZ7fhSXTprC1XTxmERpEZA/XHiFw7rIwwWIH4TmuFPPgl7O6JNPLcW756rV+tVtS7epY4/EqFvCPGSa7rOpzXOn3ltrckxhtIN5MWsYYsj8x8kqYYnlwGy2wKjJ1mkWHDGGKScI7v5uqa8p+u623e4yV7+R9xdxfqmjPFqNhohmVYWhlknjMJydhKrDmLAZIYbAhxutYNJp8GZSx5cnPFP8muN+3dV3DcovgXXet3XEnCMsOsaRDbaWiRw2tpZBla4mRlLCN32D8oYqBsPC5TkMRWvDhx4tSvcTd7tuW6SadWl2ulfO98oafdSNPotvpXDHDD3sgmu7Mkxc0qsFmbJBHhjIjIGx6nPU9hzMsc2o1CxP4a3+S+fffjgdyVWIOE76KTiG78FZjBeq0qRTY5UkBP4RnbK529hXUWknqenBGupPnyvUrclFdR9rCFb2R54+eVj0VcZ7Zr3Gk0L08VjXYwyydTshpsk2lSia1meCXJJKHHU5PxG9eiwXjaaZQ/iN/pnHNprMUdtxFZx3HKAFvI1w6++24+KkfCvSYs0ZqsiM7jXBobjhKy4g04CC5j1S2A8glIEsf+WQdD8QK3+5jNVyirqo51xDwJd8PM8yq89oNyzJh4x/iA2I/wAQ29hWDLpZR3jwWqfkSq3KASMA9z0PzqhRoawmJ2hdWUshH4WU4IPsati6AzXaBxc0LCO9G3TxlHUf4h/uPpW3HkfcraN5HDHewLLGyyRsMhl3Fb47lYFcaWhztim6UwC2XSBgqMfSp09iWK7rSJEzgYz6UHEaxe1pIjlaFBOVaXqNy0Ef3tVinI8wjOVFJ1UWUOre56eY4PpTqQA6K7VPztT2BoMh1NVbOcn3NHqANbTW3zgcuMdcZpuoA5tNTlkx+98p7Yp7IOILpEj55ZcJ3Z2wB9aF0LRGbjnT7BMRM12/YQDy/wDqO30zVc80Y+pOkz+pcdalqIKiT7jAduWA4Y/F+v0xWOWeUtlsMopC+DSrufzpD4KtvzzsI8+/m3P0qtY5vehrDouG4G3vtU5V6mO0hZyf9TcopliX9zBYxt7fRtMKtY6QJp16XOov4rA+oX8I+QFOlCP4UD5gGt3s1/g3M7SsPwp0Vfgo2FZp3P8AEwowut2aDrlmJ3NcnPiSL4sTaTZajdarJDo9oJ5+TlZnH7uIEjDO3Rdxt3O4AO9cHUey/wDyC6HdXb3r7mmOX3e5oNX4ch0rToBfuNQuolwzMCIySc/h7+2enoKxZfY+n0Sbgrb/AC+RPfvJyIbTVL/Uy2kqqCKVi33nABtYQMvy7fQnoTtvXlf/AB0cuoTiuN34+v8ANzR7xpbnmkabrPFOkavIkVxp9g8aW9t9yc27TQjPnz+J1JwN8KQDsQax6mK9nZMU2vilb338ceP1Q0X7y9wTg7gyLh5NRsLzUm05ZY3kkE0ebYQKVLhh7gMcLufc4FVanWS1ThOEeprbne3/AN7l2ydiWxm1mW1ttQ4cjstDnsL154XkYFbmN43HnV+YSEq2CGBHTpit94ITcNTc1KKTW+1NcVVK/G6oEk5bpj6PTbu9W3m4iNlpupWwTULqU3PhGPnQgu8fNhjgHY5BLYUbCsDkoSnDSRcoyuK2tbeHylf2S3Yva5AGtcSapx99neiw29o9rLHcTQyyW7sst3bnnRFk5f8A8eQ46NjJ61oxY8Ps7WZKlyk9+E1u6+u68cDtcWbbVuJ1M+kadotvaS8Sam9yIdVEaiQiOEs/iKFDOzBo1853yTvXIw6Zyhk1Gdv3cKuPbd9nfC3e3GyDfEULtF0qHhtrzUkn1a+1FNMl5VuHEksE/ISgXJ5eUcpBTcbbjflGnNn/AKqcMMklHqXycb3vw3f++4WlG2gfg7iTUtT4GTRr144LrkgkttUutlJGTyvzjlLEiUAtseZc5ODTajSY46l58UbVtOK9eGq3VbbIRzdUMtTX75wpfarZq1xZXkcTRyzBfESTAQuCoABIyrDGcrv0FDS429bDTzVOLe29VzW9/NEk7h1IU6NrcjLAZT/aYWV0mGzZByOb+L0OdyCa9tj0WP3qyJU0ZJTfTVnSLi2g1yxiuoVGHXIHcHuK9h/TrLBSRh6nF0ZuezaFyhGD6dqqWChuspU8u+MEdq0RjRLGml6vNYS80Urwt6oxGfpXRxyce5W1ZsdM+0C9wEmlWbHeRdz866MMrkipxPrkcPajlrnRo4HbrJZMYSfchcA/Smaxz/EibrgAuOFNEnyba8vbdT+WULKB89jSPBjfAeprkGPCYiP7vVI5AO0kRX+RNBadLhk6mM+HDqnDlwfDCX1m/wDeQxSZPxAOCD8OtWQhOD9APdHQLdodRtxNAxZDsVIwVPoR2Na+SvdAlxZcpyM+9QIBPBkeZc1CfICnsI5B0wagT8/KhHtWM0lityjYnPrSgLY2lA5udsfGhbIXx3L/AMWTR6mCu4VDd3CDHOAfYVOuQKQbFqd6PKtwRj5UPeSB0ky8kzgyyNMw3wxJANVuTkRpIc6Pot5rJDACG17zMOv+Udz/AMzVkMUpO3wC0a620Wz09F8Jf3g/6sm7fL0rSscY8CXZ69n4r4Ql/c0Gm+CcHzad4Y3wSaRxolkfui8pJbAHeh07bksV3iRc3LGA231NZ5KnSCLLbhS44o1DwkIitk3mn68o9F9WPbsOp6b1LD7x78B6qRsX06z4b01bWygWKJcnA3Zm/iJ7sfWrZ1jj0xQOd2cw4nd7yZ5JT5VzhewrzOpi8zL4ujQcI8EW+j6RNeapErzXIy8Ld17Ifbufjiuhp9HDBjufLElNyZl+Mry71K+aVJJLa3h2XwWKE477emdq8v7SwvM3JJelmvE62M7w1p8dz9/WeWWK4kQst0XJZUGSwJJz0329K8VrNJmSi4fElyv3NcJx39QPRtWl1Piq4n1CzX9iWcQX7rAgCgFgm2epA7k7b1kz4Y48KhjfxyfPfyWRk3J3wHnTdO1CTUra3tlvJ7yVbgyOvktVXm3bscIyqMHqB60uKGfK49Oyivv/AB7hlJJbjK+kuOBRpsGmIsVlMjRzxNKMOMfiLcrEEkg7Dfcd6yY1HWdcs/Krj9P2C5ONdIu0vxuI72DUbOKPT+J9MSPwLtWJW4K5V33AI5gzZXp5hjvXcWifTLFDeEr+H0fZfLYqU97fI+seM34PjtND8EXLyeJOl1dHlLtJM7GNlx0LO6jBGBjrvXFei/rcnvnaapUv/wApL7ruXOfQqL9E4ck1Ow1nT5ZjcRW0MSxRlRyoF5tgPgAMewr1PsbRyzxyZHGnyZc0+lpCmzsryws5tKM0kdpzmTws+UnIOcfIV6GHs3H71Z3H4/Pcz+9bXSKhaG2uW5tjntW94uli2bPg3XV065Frcyf2Sc+Vz0Rux+B6fT3rsaOfQ+mXDKci7o2etaB4i8wXfqCK6k8KbtFKZkbzTZbdz3FZ/dtD2CBW5jtuKKi0G+xZHIw6/ptVy9AMZ2+pPGAreZP1FaVa5EoOjuyR5TViAWpfkjzj55p0yF0d0QQVY9adOiGj0HXZIJMM+QdjmrUwM1olSdAw2J9O9F8Cgk8IIJGx9qVNPclC2eIqc/XG1NRDgDWnP16VlcS6z1LHfIHTuaHSGyQsGbqeYelLQLLFs+XqOWhRGTSHlGQOYeppaJZdFGXcALzsSAFXuewxQUbYbNtoHBix8tzqXKx6rbA7D/N6/D/2rVDEo7yK2zULOwTkjUBBsMDoKvsUlFaeO2WBHqaWiF5tliAVGxt8zUoBBrORk5mYAZ2AquSfcgDdYUAcpZvTO1VyYRWllLqN4ttAoDucFj0X3+AqhLqlSG4N3aafFpNiltbgLGgyWbqSerH3NaXUVSE5Mlrt595mdY2JXGOY9veuZkl1NpDoXcP8OR6le/e51/slsebf879h8tj9KTBhi31Pgjdmi1CM3x2UsBsqAbAVdk+JgRkeKOHj4arygc25GK5uowKS3LIypmf0PRlju7jy8qmN1z6eU5rix0yk5fJl3XwYHhS08G1vZpebDWvNy5zk5U/XevOf0NR6e5r6/sdT0jg5tG4VYSxiPUL5OeYd0BGyfBR19y1ej03slaTTe7e8ny/54M08znK+wg1jR7j7qokYyiNSqh9wo9vTpXHyex8XV1QVb267lqzPhizQbX7re+IvlZDke471ux6XpkmhZS2C/tK0QXtlaatEoK5FvcLjuclG+fmB9wPWtGb2dj95/VQVN8+Pn8xY5LXQzVfYpqS6rqd5a3ZzeNaY5m6yqrAc3uQGwfXr610vZuFRnJeUV5XdGg4p4TWOZpETK46gdK6jwpMps5zqOlvBMzcvMoyDWSeF2WqSFVuiiUpvj3qmMKY7dm/4M4uFsqaZqrc1qTyxXDdYv8Lf4fQ9vhjHYwZf7ZGeUe6NVq/D6lWIGVNbHASzH6hobQnK/Wk92EVyW5RwCevQ0vQ0xrPgjK2CckUyTWwC6CQIMEGnSAGo7HJBJGKdELFbJ605AmCblAwSp+NMmQ0ujcQvCBFM3MnTerEwUOTcSqviRSeNEezdV+fel6adoB6k7TdV7dqawHGEs1OORR7k1XQ56bbBwQD7AUrIeOOXI5QPegwlZAb8o+JqtjFfh8zBVQsxO2O5oVb2Abbh7QxpKC4nUNeMOnaIeg9/U1qjHpQjdjxHklO65HwqwFh9nbsc5XA9aBGwvPm5FUD2HWoQ+az5jkkj0ANAVg146xpyqxZunwFUzYd+4pnuuVdhhR1Y1S34CP8AhvSjbQ/e5V5Zpl2U/lTtn3PU/IdquiulWB7n2t3pKvGmcY3rLmltSCjPW+mSalOIF2By0j4zyjuf+dcis+PH1DcGmttOSKBLZFKRRjHIDnHt7nuT3JPbFbXGl0oWxktkqxjlQLt2paSVEsQ69pvPliBv5RntWLLGxkYvUbf7jo+rzKMMlrIqnvzOPDH6vWJ41CEmOnbFH2ccLR6lxFbwNGGgjj8aUY25VIwD8W5B9ay6LTrJmtrZFk5UjsWpaN94Y5UbjfNegnj6jPZnNZ4cSSywEyQD261lnhTXAUznc+kNZ3S+XrWN4aZZ1DnSIrfUYbjS7xuS3ukMDSfwZ/C4/wArcrfKtEMdrpYt90Y/he5m4O4ksrueIwz2E7Q3UQG6jdJV+WSR/lFY8K9xlTZdL4kfo6eCHUrRZY+WRHUMGA6j1r0DV7mWzmPEHDbWssxVcr1wR2rO8fI1nPL3TBHMSAV9PSsUsW9liltRVbuxU5GR03oVuFs6DwLxaFWPSdQf90cLbyv+U9kJ9PT6eldHDkv4ZFTXg1Wo6YrKwZMrv061tcU0VpmV1XROWNiF5l65FVuOw9iGW38JVfJZO/tSNBPOUKwK7g96lUQvhJDbbUSF7QsAH7e1OQ8V8ddqhAqG5C4Vl+dOmQcaZrDWpwDzJ3U09gNDBqEUgDr+A9h2qNEo5MMlc8x+ApAkWmRCDnHwoEKXkJ3AGPU0jGIqjP1Yk+gpelkNFw3pYh/tkvmfP7pcdPVv6VfCNbitmliWWb8o3q0UY2kch3wMZ7d6ABnGXVeXFFtELII3U5CDbcn3oWArurwxK2Fznq1UykET3dy7LylAq/qaqfgJ5pdn+0r1FdAIIxzuPX0HzP6Zoxim7IzRXl5yITkdOnrRnICRlru7ZmON87k1z5S7sc0ug2DWlqgPluJj4jeqjtn4Z29yT2FbcMHCC6uWVt2OYrTYxxLgt1YjoKZrekEYy2qpDyqNx3oyjSIZniNSPDQAYUEn49qyZN3Qy2MLxRHy6QkIIzcTg/6UGf8A/TL9Kw6jbFXkePJoPsc0cINUvCvmJSBf8oBY/qV+laPZ+NKMpeRcjs6DcWvMuPwkjrXTcbETFc1sjIsJGebIzVTj2ImYHXtI8GdxjJXp71Q4UhrMjGxgn5iMDOCKoiEs4805buGy16Igi6xa3g9J1Xysf86DPxU+tLqMd1NDwfY3P2PcQi80qTSZ3BubLzRg9XhPTHrg7H/T61s00uqHS+UJNUzWa7oy3URZBlsbZrQ4iWcu1jQgY5VMeD2OO9UOC7jJmLu9NaJGIGGzvgVncHVjWUoxKAMuD3NLwE6TwVxaupRppl+/9oAxDMT/AHg/hPv/AD+PXfiyWqZW4j28thCxBGx7VpFRm7/SUdHeFckblPX4Ujj4HsQz2C82YyRkdKr6Qg8bGNir7EHoe9LQRjbsjgeY/DtTAJPa5BwKhCgoyH2qcELIpCh26U1kD7TUXiyQdvSrLIYU3gPf5A1WNwR8cPjlG9D5ALEUE5c1ADjRrEX8pYj+zod+wY+gqyMe5DWwRDpkKOg2q0UaW1v0J3HoO9EgeJhkAnCetCkLYZHOqqQuR2O3Wg2Qi1y6KcDA9TS2EVX1+/Nv5QN6qbRBZeXZdSzZVQNjVLduwjzRh92sE5jyyy/vHJ2wOw+n86tukABvb7xHkOfKOmazSdjI+0qyN7NHM6c0St5VK/jYf7CmjiTakwSdG1sbNubz5LNuxNaqbZWN7W2EYJz5uvwoVQS6cCOAt6ClnwRGI1aXxVYk7nqTWdrYcxnETiXVGhI8lnEsR36yEc7/AELY/wBNYM/xOvBYtjo/2aWX3DhmFmGGmdpCfXJ/7V09LDoxJFUnbNLcocqQua08C2A/dvGhLA+bPlzSVsSzP67pnjocjz9M+tJ02G0jm+v6a0UrALgL+Ij1qiUEOiGmtHeWl3ply2LS8QIzHfkcENHIPdWA+RYd6nSmnEK23M3Z3d9wxrayxfur2zkKsvZuzKfY/wC+axRbxTH/ABH6G4f1eDiPRoL63PNHMuSp6q3QqfcHauxFqSsoa3EmuaOoeQMPK+6+xoOKZLtGI1Th9ZYmZVKsNsH19KrcNg2ZDUNGkt151Q8vXNUuI1i4Fom5gCrDcEbGkquAnSeFOKY+ILb7ldvi/jXyudvFHr8R3+tbsc+pUxGj28R7WZgr4x69DVjTRBZcrFcEE/u5e+1VNrqp8jIV3trySZdRv+cd6DQSqJPBfY8y/wAqFEDoZ98evY0xCySESLtUZAORDGd9qQhFJOXruKKdEOaHWY1PXNCwkH1+GEhpJkjA7uwH86XqoAXovEFrrWoxWVpcpPO+55PMFUdWJGwA/oO9OviewdjqFgqRRRxR7RqMD3960IVjmFUwCelEUOU85Rm8qVLJ8gwSooGFIx60pD1rtXIBPkXp70CHksi5LlsgdAelKQU3l0n3jff0FVPdhBwheYSSrhB1Un9KVK9wh15flVGOp657GhKRKoEs7V9QuliBKx455GHYVXjh1Pcl0bfTbZVC8qBVUAKB0ArdVFY4gRi3L1B6ihZBssZWPPT1oMgLqsyrCFHc9qWfFBRjrnle4BbAQHmbPoOtZ2FMwF1K8yTTsDzzSNK3xJLf74rmS3l9S1HbtAtBaaPYwEYKRKPnjeu/GOyRQ+Qi5mLMo+o9RT0iFMbZDfHpVdABTD4kjq/Y8wz70QmN4o0gxSSyKvMh6mqnHuFMwk9o0EzMuOUHGRVTjTHTsF4nt/vUdrqaDzOPu8/+ZR5G+a7f6fes2aDdSHi+xovsf4i/Z+syaXKcQXY54weiyDr9R/KrdPP+0WS7nXNRsVuoGU7elb6KjH3unuokCtkEYO360KIZ79ms0UiuOYA4OKSg2INR4c58Mq8p77VW4JhTM3cadc6ZcB0LIUYMrr1U9jmquHSHNVpvFMerwrb3oEV2Ng/RZP6H271qjO1TFqge+zHIOuRtnuKWcbCgb76SpR/OPQ1FxuE8Kq3mjOD3FMQtjkD/AIlIb1oUQIUHl6ZqUQ+cCRd/rSvcgvlhKtt0NJ3Ifgk8bapNs2o3TD/9zf1rA5vyQI0/VZbmYF3aQ+p3NBXJ0Q/UX2W6CvDmkxiRc39yA9w38Hog+Hf3z7V1McVFUiHWLJ1jQewzVtgGkNxzFcbnO2KgBh98zu/5RttUIe/eQ8uWzjsBShotUq7b536CgCiu9mCBQOp6UGSgJ4+Q88hHN12pKCfCVUDSPvjcexpW6IgBrzxXLHf2qnkJr+HdKNtbhWB8RzzyH37D4D+tbYQ6UIzVxRLEFydh2FWNbijGIBRkDBO9LVEC+dQjHO460GQR6vOrR82arku4UZLVJiLO4cbFhyD57fyzWeWyYUZWSES3EMC7B5Uj+rAf71jgviLDtlu2WwBt/Ku4tilkLwhyoQ4YZoSTZCm0cM7EjfbIoehAl4v37Y32oNEFGr2w8No3XIPQ+tTsQ51rOmGzlflwY29fWqmhkKWt/vWm3tty7lOZf8y7g/pSyjaaGRlbO7e0uobuI8skLrIpHt/WufF9MrH5R+lNH1KPWNKgukbKyICSPWuxF2ilgOp2vhSeKv4SNxRqgGemtwpJU49R60pCpYVcOCOnUVKCDXOhwXCkMoKt2xQ6Fdksy+p8FKsh5RjuPelcNxkxf+yLlF8MXHOB0Em+PnUoNiu7gubZvNGS3qveq2+nkPIPDfsW5c8pHUdCKKkGg+K+HRqZMFBsVwOqt9aJC0TKwIIwaFohFlDd/rQoh/NuOBgwri2Q13Atp4uv6er7r4ykj4HNaMXNkP1fw1KFjTffI3NdGLohuLOYy8ox1qyyDeG6WH8A83apYKL4bjxM5bLH0qBCoCM9SfXeoQl45RiFJY5+OKjIVOwQDLczk9epoEBJ7gNMqc253x6CksANfXvMwjXfHWqZBDeF7P8AaN/4hGYYN/Zm7D/f6U+KNuwNnR7GEKvv1rYisYQqCydz1pXyQYx79Bj41GQkZQsTD61HxYV4M3qrnkbJ8uarZDL6tIPCjT+Jix+A/wDes0+KHQjtXVtc08YypuI9v9QqmEamgnYLKbK5z5fWusvDKiu6cGVWBIOCMj/nxqcEK7CbnkLKRs24+dVryQbMvJO5G2RnFOtyFN/GtzbrlQGIyMDpUcbIYXXbRYyysuUz9M1WMjP2Vusd/wArbj+YoUGzn19bGzvrmAjBikZPocVy5rpkWI619jWs+PYT6c5y0J5lz6f8/lW/DK4iSN9PGGypGRjpWwrMvd8sNw0Lryg7q3tVbVBQKp5SRgFh2HpURGFRKlxECNvUEdKf5AIXVn40XL+dehpiGa1KyKODjHN3HY0jQyEd5ELiFo22kHQnsRVUlaoZCV7dLtMOB4g236isjVFgHJavCCR27Hej1MNFS3Txds06n5BRfHq6gHI6e1HrQOkk2sooGATU6kTpPwWtqpYf0riWKangtY7XW7KRyAqyDf8AT/etOKW4T9F8Py55VzsK3RYzRtbO6xGArfOr06FGdnM7McH50wBpAQoHmFGiBiSllyCAvr60aITEgxhQAcfrUZBTq+sQaWnM58SfHliB6/0FUzmooKVibS76SZZ72dsyynlUDoqjsPmT9KpjJtWwtHz3mQzE+wFDkh03hXTRp+nxIwxIRzyH1Y1vhGkVs08ZOeUEAd6dt9hUkHQ4iCgbnqSaT5kCzIVdObAB3pu4D6eXHMBsMUoTL6vcgK2O5xVb4IjMalIWdm68q8oFZZu2OhDDd+FrVi5wMXEYB/1Clj+JMJ17TpiYs5GP5V00VEppuVOU9s1GrJYNpE2JXUHOCaT0DRoYpMyhm3IGwqyuwCy5TEQYnr2qAMprkYYMGXmjZRn2waRrcYxc0XgXKsGyqnGaQYxfFScvEN2TuHYOP9Sg/wBa5+ZfGWLga/ZpqR07jC2UthJ8wnfrkZH8v1qzDKnRJLazu0+GjBPXsRXSW5QJNYiFxb+Ko8y57fpQohnfHXyyIe+GBpGMM7SZZcMpwfzA9/emSI9w5oyMHt0p0KK9TtQxI5ch9vhUqwoxmrQGB2JGMdc1XJUMjPXLqlyD+Hn7+9Zci3ssRYUEo3xn1qoIBc2mx7GpQwve3KtkfCo0QqEORsfippWiH4tCgmuUVDTTV3wO9WR2Yx+gOF51urC3mjJ5HQEc3UY2OfmDW+LLDb2G6A52xWpCsdWrscAfHNWIWhnBgBc5JPSmAHRnmUljyooyfYUQGd1PitnUpYAxqf8ArON/kO3zrJPLWyHUfJlp5HldnZi7HcsxyTWNtstqthvC5itIox0C1o42K2MOHrUajrtrE/4FbnI9l3/pVmNXIVnYbVfLmtqKw1TzN8qgBlFnkUjG2KHcHYIJ25s5IoDFV2S2MHc7Uz2FRltWOJVXt1NVMJmrliUcnq2TWR7jmVvpCl1akE58dTkezCl4aCjsWmTESSRncCuouCqid2xLq4JBBI+IoPYNAukTc11L1DAn+dKgM1dvIWMbHqCacFB0/nhz26YokM3qYyjHGQObIqEMLeoFmkUHyMcgelVscxXFRzqme5iX9Mj/AGrFmW5YhVZTyWdxDco2HicOPiDVGNtbhas/R+m6gNQ0y2ugMLKgYj4iuvHgpa3KbhgpJ7E4PxpqBRj9diWxu1lQYWQ4ZR61W7GR7ZXZinB3KHqKiZKNNbScxCnLKemafkVojNGJQwI6U3Yhk9etl8J2IyRsfel5GRgbqLxOZf4DgGs0x0U2908J5Tkj0qgZbjEBZ0z7VAgc9upJwN6gQJoOV8gb1CH/2Q==\"\n", + " }\n", + " ]\n", + " ],\n", + " \"parameters\": {\n", + " \"maxPredictions\": 2.0,\n", + " \"confidenceThreshold\": 0.5\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "h6IskqWe_jAo" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_predict:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"prediction\"].predict(\n", + " endpoint=endpoint_id,\n", + " instances=instances,\n", + " parameters=parameters,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "KHf2BSMR_jAo" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "egOqdNyN_jAp" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_predict:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"predictions\": [\n", + " {\n", + " \"confidences\": [\n", + " 0.999991059\n", + " ],\n", + " \"ids\": [\n", + " \"5133242657597816832\"\n", + " ],\n", + " \"displayNames\": [\n", + " \"daisy\"\n", + " ]\n", + " }\n", + " ],\n", + " \"deployedModelId\": \"5165312113245159424\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new" + }, + "source": [ + "### [projects.locations.endpoints.undeployModel](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/undeployModel)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "KrVZz6Uw_jAp" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_undeploymodel:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint_id,\n", + " deployed_model_id=deployed_model_id,\n", + " traffic_split={},\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wvFK-kir_jAq" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "GvajSw-Y_jAq" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "MPWmXvPVUIL2" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_and_export_edge_model:migration" + }, + "source": [ + "## Train and export an Edge model\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "5bwEQMKT_jAr" + }, + "source": [ + "### [projects.locations.trainingPipelines.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.trainingPipelines/create)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "LOsxiKj4_jAs" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_create:migration,new,request,icn,edge" + }, + "outputs": [], + "source": [ + "task = Value(\n", + " struct_value=Struct(\n", + " fields={\n", + " \"disable_early_stopping\": Value(bool_value=False),\n", + " \"model_type\": Value(string_value=\"MOBILE_TF_VERSATILE_1\"),\n", + " \"multi_label\": Value(bool_value=False),\n", + " \"budget_milli_node_hours\": Value(number_value=8000),\n", + " }\n", + " )\n", + ")\n", + "\n", + "training_pipeline = {\n", + " \"display_name\": \"flowers_edge_\" + TIMESTAMP,\n", + " \"input_data_config\": {\n", + " \"dataset_id\": dataset_short_id,\n", + " \"fraction_split\": {\n", + " \"training_fraction\": 0.8,\n", + " \"validation_fraction\": 0.1,\n", + " \"test_fraction\": 0.1,\n", + " },\n", + " },\n", + " \"model_to_upload\": {\n", + " \"display_name\": \"flowers_edge_\" + TIMESTAMP,\n", + " },\n", + " \"training_task_definition\": TRAINING_SCHEMA,\n", + " \"training_task_inputs\": task,\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateTrainingPipelineRequest(\n", + " parent=PARENT,\n", + " training_pipeline=training_pipeline,\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "IREmlyf5_jAs" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"trainingPipeline\": {\n", + " \"displayName\": \"flowers_edge_20210226014942\",\n", + " \"inputDataConfig\": {\n", + " \"datasetId\": \"3094342379910463488\",\n", + " \"fractionSplit\": {\n", + " \"trainingFraction\": 0.8,\n", + " \"validationFraction\": 0.1,\n", + " \"testFraction\": 0.1\n", + " }\n", + " },\n", + " \"trainingTaskDefinition\": \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_image_classification_1.0.0.yaml\",\n", + " \"trainingTaskInputs\": {\n", + " \"multi_label\": false,\n", + " \"disable_early_stopping\": false,\n", + " \"budget_milli_node_hours\": 8000.0,\n", + " \"model_type\": \"MOBILE_TF_VERSATILE_1\"\n", + " },\n", + " \"modelToUpload\": {\n", + " \"displayName\": \"flowers_edge_20210226014942\"\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "PWMEUCbF_jAt" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "T_8-j8nx_jAt" + }, + "outputs": [], + "source": [ + "request = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT,\n", + " training_pipeline=training_pipeline,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "OalQ6m9P_jAu" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "X6gz-Nx8_jAu" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "trainingpipelines_create:migration,new,response,icn,edge" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/trainingPipelines/7472017139575029760\",\n", + " \"displayName\": \"flowers_edge_20210226014942\",\n", + " \"inputDataConfig\": {\n", + " \"datasetId\": \"3094342379910463488\",\n", + " \"fractionSplit\": {\n", + " \"trainingFraction\": 0.8,\n", + " \"validationFraction\": 0.1,\n", + " \"testFraction\": 0.1\n", + " }\n", + " },\n", + " \"trainingTaskDefinition\": \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_image_classification_1.0.0.yaml\",\n", + " \"trainingTaskInputs\": {\n", + " \"modelType\": \"MOBILE_TF_VERSATILE_1\",\n", + " \"budgetMilliNodeHours\": \"8000\"\n", + " },\n", + " \"modelToUpload\": {\n", + " \"displayName\": \"flowers_edge_20210226014942\"\n", + " },\n", + " \"state\": \"PIPELINE_STATE_PENDING\",\n", + " \"createTime\": \"2021-02-26T03:34:33.436419Z\",\n", + " \"updateTime\": \"2021-02-26T03:34:33.436419Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "QidD3YK5_jAv" + }, + "outputs": [], + "source": [ + "# The full unique ID for the training pipeline\n", + "training_pipeline_edge_id = request.name\n", + "# The short numeric ID for the training pipeline\n", + "training_pipeline_edge_short_id = training_pipeline_edge_id.split(\"/\")[-1]\n", + "\n", + "print(training_pipeline_edge_id)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "vukYRydw_jAv" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=training_pipeline_edge_id)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " break\n", + " else:\n", + " model_edge_id = response.model_to_upload.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(model_edge_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "models_export:migration,new" + }, + "source": [ + "### [projects.locations.models.export](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.models/export)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "lqJoqYMI_jAv" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "models_export:migration,new,request,edge" + }, + "outputs": [], + "source": [ + "model_output_config = {\n", + " \"export_format_id\": \"tflite\",\n", + " \"artifact_destination\": {\n", + " \"output_uri_prefix\": \"gs://\" + f\"{BUCKET_NAME}/export/\",\n", + " },\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.ExportModelRequest(\n", + " name=model_edge_id,\n", + " output_config=model_output_config,\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "WRi2WK4o_jAw" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/1204871229996007424\",\n", + " \"outputConfig\": {\n", + " \"exportFormatId\": \"tflite\",\n", + " \"artifactDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210226014942/export/\"\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "v6isqzPQ_jAw" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "models_export:migration,new,call,edge" + }, + "outputs": [], + "source": [ + "request = clients[\"model\"].export_model(\n", + " name=model_edge_id,\n", + " output_config=model_output_config,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ZCyd1qAb_jAx" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "xRGA3tcC_jAx" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "models_export:migration,new,response,edge" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "nQAiuIAc_jAx" + }, + "outputs": [], + "source": [ + "model_export_dir = model_output_config[\"artifact_destination\"][\"output_uri_prefix\"]\n", + "\n", + "! gsutil ls -r $model_export_dir" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "fCG8yqkW_jAy" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210226014942/export/:\n", + "\n", + "gs://migration-ucaip-trainingaip-20210226014942/export/model-1204871229996007424/:\n", + "\n", + "gs://migration-ucaip-trainingaip-20210226014942/export/model-1204871229996007424/tflite/:\n", + "\n", + "gs://migration-ucaip-trainingaip-20210226014942/export/model-1204871229996007424/tflite/2021-02-26T04:43:08.209439Z/:\n", + "gs://migration-ucaip-trainingaip-20210226014942/export/model-1204871229996007424/tflite/2021-02-26T04:43:08.209439Z/model.tflite\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup:migration,new" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aoJ18d8Y_jAy" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_pipeline = True\n", + "delete_batchjob = True\n", + "delete_edge_model = True\n", + "delete_edge_pipeline = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex AI fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset:\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex AI fully qualified identifier for the model\n", + "try:\n", + " if delete_model:\n", + " clients[\"model\"].delete_model(name=model_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex AI fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint:\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex AI fully qualified identifier for the training pipeline\n", + "try:\n", + " if delete_pipeline:\n", + " clients[\"pipeline\"].delete_training_pipeline(name=training_pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex AI fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob:\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the Edge model using the Vertex AI fully qualified identifier for the Edge model\n", + "try:\n", + " if delete_edge_model:\n", + " clients[\"model\"].delete_model(name=model_edge_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the Edge training pipeline using the Vertex AI fully qualified identifier for the Edge training pipeline\n", + "try:\n", + " if delete_edge_pipeline:\n", + " clients[\"pipeline\"].delete_training_pipeline(name=training_pipeline_edge_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r gs://$BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "collapsed_sections": [ + "gcp_authenticate", + "bucket:batch_prediction", + "setup_vars", + "import_aip", + "aip_constants", + "automl_constants:automl" + ], + "name": "UJ1 unified AutoML Vision Image Classification.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/migration/UJ10 legacy Custom Training Prebuilt Container SKLearn.ipynb b/notebooks/community/migration/UJ10 legacy Custom Training Prebuilt Container SKLearn.ipynb new file mode 100644 index 000000000..f49bfd55f --- /dev/null +++ b/notebooks/community/migration/UJ10 legacy Custom Training Prebuilt Container SKLearn.ipynb @@ -0,0 +1,1942 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title:migration,new" + }, + "source": [ + "# Vertex SDK: Train and deploy an SKLearn model with pre-built containers (formerly hosted runtimes)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the Google *cloud-storage* library as well.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-storage" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the Kernel\n", + "\n", + "Once you've installed the Vertex SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "if not os.getenv(\"AUTORUN\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU run-time\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your GCP project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a GCP project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebooks.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend when possible, to choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You cannot use a Multi-Regional Storage bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see [Region support for Vertex services]()\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your GCP account\n", + "\n", + "**If you are using Google Cloud Notebooks**, your environment is already\n", + "authenticated. Skip this step.\n", + "\n", + "*Note: If you are on an Vertex notebook and run the cell, the cell knows to skip executing the authentication steps.*\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your Google Cloud account. This provides access\n", + "# to your Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Vertex, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this tutorial in a notebook locally, replace the string\n", + " # below with the path to your service account key and run this cell to\n", + " # authenticate your Google Cloud account.\n", + " else:\n", + " %env GOOGLE_APPLICATION_CREDENTIALS your_path_to_credentials.json\n", + "\n", + " # Log in to your account on Google Cloud\n", + " ! gcloud auth login" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n", + " BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip" + }, + "source": [ + "#### Import Vertex SDK\n", + "\n", + "Import the Vertex SDK into our Python environment.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "04c9abd7432f" + }, + "outputs": [], + "source": [ + "import json\n", + "import os\n", + "import sys\n", + "import time\n", + "\n", + "from googleapiclient import discovery" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `PARENT`: The Vertex location root path for dataset, model and endpoint resources.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients" + }, + "source": [ + "## Clients\n", + "\n", + "The Vertex SDK works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the server (Vertex).\n", + "\n", + "You will use several clients in this tutorial, so set them all up upfront.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients" + }, + "outputs": [], + "source": [ + "client = discovery.build(\"ml\", \"v1\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "36dd422ee146" + }, + "source": [ + "## Prepare a trainer script" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "5b0c2221146a" + }, + "source": [ + "### Package assembly" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "6d54d3711c9f" + }, + "outputs": [], + "source": [ + "# Make folder for python training script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\\n", + "tag_build =\\n\\\n", + "tag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\\n", + "setuptools.setup(\\n\\\n", + " install_requires=[\\n\\\n", + " ],\\n\\\n", + " packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\\n", + "Name: Custom Census Income\\n\\\n", + "Version: 0.0.0\\n\\\n", + "Summary: Demonstration training script\\n\\\n", + "Home-page: www.google.com\\n\\\n", + "Author: Google\\n\\\n", + "Author-email: aferlitsch@google.com\\n\\\n", + "License: Public\\n\\\n", + "Description: Demo\\n\\\n", + "Platform: Vertex AI\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "5e87d510414c" + }, + "source": [ + "### Task.py contents" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "077d9111ff59" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Single Instance Training for Census Income\n", + "\n", + "from sklearn.ensemble import RandomForestClassifier\n", + "import joblib\n", + "from sklearn.feature_selection import SelectKBest\n", + "from sklearn.pipeline import FeatureUnion\n", + "from sklearn.pipeline import Pipeline\n", + "from sklearn.preprocessing import LabelBinarizer\n", + "import datetime\n", + "import pandas as pd\n", + "\n", + "from google.cloud import storage\n", + "\n", + "import numpy as np\n", + "import argparse\n", + "import os\n", + "import sys\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv('AIP_MODEL_DIR'), type=str, help='Model dir.')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "\n", + "# Public bucket holding the census data\n", + "bucket = storage.Client().bucket('cloud-samples-data')\n", + "\n", + "# Path to the data inside the public bucket\n", + "blob = bucket.blob('ai-platform/sklearn/census_data/adult.data')\n", + "# Download the data\n", + "blob.download_to_filename('adult.data')\n", + "\n", + "# Define the format of your input data including unused columns (These are the columns from the census data files)\n", + "COLUMNS = (\n", + " 'age',\n", + " 'workclass',\n", + " 'fnlwgt',\n", + " 'education',\n", + " 'education-num',\n", + " 'marital-status',\n", + " 'occupation',\n", + " 'relationship',\n", + " 'race',\n", + " 'sex',\n", + " 'capital-gain',\n", + " 'capital-loss',\n", + " 'hours-per-week',\n", + " 'native-country',\n", + " 'income-level'\n", + ")\n", + "\n", + "# Categorical columns are columns that need to be turned into a numerical value to be used by scikit-learn\n", + "CATEGORICAL_COLUMNS = (\n", + " 'workclass',\n", + " 'education',\n", + " 'marital-status',\n", + " 'occupation',\n", + " 'relationship',\n", + " 'race',\n", + " 'sex',\n", + " 'native-country'\n", + ")\n", + "\n", + "\n", + "# Load the training census dataset\n", + "with open('./adult.data', 'r') as train_data:\n", + " raw_training_data = pd.read_csv(train_data, header=None, names=COLUMNS)\n", + "\n", + "# Remove the column we are trying to predict ('income-level') from our features list\n", + "# Convert the Dataframe to a lists of lists\n", + "train_features = raw_training_data.drop('income-level', axis=1).values.tolist()\n", + "# Create our training labels list, convert the Dataframe to a lists of lists\n", + "train_labels = (raw_training_data['income-level'] == ' >50K').values.tolist()\n", + "\n", + "# Since the census data set has categorical features, we need to convert\n", + "# them to numerical values. We'll use a list of pipelines to convert each\n", + "# categorical column and then use FeatureUnion to combine them before calling\n", + "# the RandomForestClassifier.\n", + "categorical_pipelines = []\n", + "\n", + "# Each categorical column needs to be extracted individually and converted to a numerical value.\n", + "# To do this, each categorical column will use a pipeline that extracts one feature column via\n", + "# SelectKBest(k=1) and a LabelBinarizer() to convert the categorical value to a numerical one.\n", + "# A scores array (created below) will select and extract the feature column. The scores array is\n", + "# created by iterating over the COLUMNS and checking if it is a CATEGORICAL_COLUMN.\n", + "for i, col in enumerate(COLUMNS[:-1]):\n", + " if col in CATEGORICAL_COLUMNS:\n", + " # Create a scores array to get the individual categorical column.\n", + " # Example:\n", + " # data = [39, 'State-gov', 77516, 'Bachelors', 13, 'Never-married', 'Adm-clerical',\n", + " # 'Not-in-family', 'White', 'Male', 2174, 0, 40, 'United-States']\n", + " # scores = [0, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]\n", + " #\n", + " # Returns: [['State-gov']]\n", + " # Build the scores array.\n", + " scores = [0] * len(COLUMNS[:-1])\n", + " # This column is the categorical column we want to extract.\n", + " scores[i] = 1\n", + " skb = SelectKBest(k=1)\n", + " skb.scores_ = scores\n", + " # Convert the categorical column to a numerical value\n", + " lbn = LabelBinarizer()\n", + " r = skb.transform(train_features)\n", + " lbn.fit(r)\n", + " # Create the pipeline to extract the categorical feature\n", + " categorical_pipelines.append(\n", + " ('categorical-{}'.format(i), Pipeline([\n", + " ('SKB-{}'.format(i), skb),\n", + " ('LBN-{}'.format(i), lbn)])))\n", + " \n", + "# Create pipeline to extract the numerical features\n", + "skb = SelectKBest(k=6)\n", + "# From COLUMNS use the features that are numerical\n", + "skb.scores_ = [1, 0, 1, 0, 1, 0, 0, 0, 0, 0, 1, 1, 1, 0]\n", + "categorical_pipelines.append(('numerical', skb))\n", + "\n", + "# Combine all the features using FeatureUnion\n", + "preprocess = FeatureUnion(categorical_pipelines)\n", + "\n", + "# Create the classifier\n", + "classifier = RandomForestClassifier()\n", + "\n", + "# Transform the features and fit them to the classifier\n", + "classifier.fit(preprocess.transform(train_features), train_labels)\n", + "\n", + "# Create the overall model as a single pipeline\n", + "pipeline = Pipeline([\n", + " ('union', preprocess),\n", + " ('classifier', classifier)\n", + "])\n", + "\n", + "# Split path into bucket and subdirectory\n", + "bucket = args.model_dir.split('/')[2]\n", + "subdir = args.model_dir.split('/')[-1]\n", + "\n", + "# Write model to a local file\n", + "joblib.dump(pipeline, 'model.joblib')\n", + "\n", + "# Upload the model to GCS\n", + "bucket = storage.Client().bucket(bucket)\n", + "blob = bucket.blob(subdir + '/model.joblib')\n", + "blob.upload_from_filename('model.joblib')\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0460d3d23907" + }, + "source": [ + "### Store training script on your Cloud Storage bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "c2eb48cc2368" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz gs://$BUCKET_NAME/census.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_create_and_deploy_model:migration" + }, + "source": [ + "## Train a model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0oqIBOSnJjkW" + }, + "source": [ + "### [projects.jobs.create](https://cloud.google.com/ai-platform/training/docs/reference/rest/v1/projects.jobs/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "1aafbc0ce6b7" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "248f10285c0a" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_job_SKL\" + TIMESTAMP\n", + "\n", + "training_input = {\n", + " \"scaleTier\": \"BASIC\",\n", + " \"packageUris\": [\"gs://\" + BUCKET_NAME + \"/census.tar.gz\"],\n", + " \"pythonModule\": \"trainer.task\",\n", + " \"args\": [\"--model-dir=\" + \"gs://{}/{}\".format(BUCKET_NAME, JOB_NAME)],\n", + " \"region\": REGION,\n", + " \"runtimeVersion\": \"2.4\",\n", + " \"pythonVersion\": \"3.7\",\n", + "}\n", + "\n", + "body = {\"jobId\": JOB_NAME, \"trainingInput\": training_input}\n", + "\n", + "request = client.projects().jobs().create(parent=\"projects/\" + PROJECT_ID)\n", + "request.body = body\n", + "\n", + "print(json.dumps(json.loads(request.to_json()), indent=2))\n", + "\n", + "request = client.projects().jobs().create(parent=\"projects/\" + PROJECT_ID, body=body)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"uri\": \"https://ml.googleapis.com/v1/projects/migration-ucaip-training/jobs?alt=json\",\n", + " \"method\": \"POST\",\n", + " \"body\": {\n", + " \"jobId\": \"custom_job_SKL20210302140139\",\n", + " \"trainingInput\": {\n", + " \"scaleTier\": \"BASIC\",\n", + " \"packageUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210302140139/census.tar.gz\"\n", + " ],\n", + " \"pythonModule\": \"trainer.task\",\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210302140139/custom_job_SKL20210302140139\"\n", + " ],\n", + " \"region\": \"us-central1\",\n", + " \"runtimeVersion\": \"2.4\",\n", + " \"pythonVersion\": \"3.7\"\n", + " }\n", + " },\n", + " \"headers\": {\n", + " \"accept\": \"application/json\",\n", + " \"accept-encoding\": \"gzip, deflate\",\n", + " \"user-agent\": \"(gzip)\",\n", + " \"x-goog-api-client\": \"gdcl/1.12.8 gl-python/3.7.8\"\n", + " },\n", + " \"methodId\": \"ml.projects.jobs.create\",\n", + " \"resumable\": null,\n", + " \"response_callbacks\": [],\n", + " \"_in_error_state\": false,\n", + " \"body_size\": 0,\n", + " \"resumable_uri\": null,\n", + " \"resumable_progress\": 0\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "5cd15f96178f" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "8a8886c9b042" + }, + "outputs": [], + "source": [ + "result = request.execute()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "a593859b87c2" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "f4d93536746e" + }, + "outputs": [], + "source": [ + "print(json.dumps(result, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"jobId\": \"custom_job_SKL20210302140139\",\n", + " \"trainingInput\": {\n", + " \"packageUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210302140139/census.tar.gz\"\n", + " ],\n", + " \"pythonModule\": \"trainer.task\",\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210302140139/custom_job_SKL20210302140139\"\n", + " ],\n", + " \"region\": \"us-central1\",\n", + " \"runtimeVersion\": \"2.4\",\n", + " \"pythonVersion\": \"3.7\"\n", + " },\n", + " \"createTime\": \"2021-03-02T14:09:33Z\",\n", + " \"state\": \"QUEUED\",\n", + " \"trainingOutput\": {},\n", + " \"etag\": \"YQ/mo0C8EUg=\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "training_pipeline_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The short numeric ID for the custom training job\n", + "custom_training_short_id = result[\"jobId\"]\n", + "# The full unique ID for the custom training job\n", + "custom_training_id = \"projects/\" + PROJECT_ID + \"/jobs/\" + result[\"jobId\"]\n", + "\n", + "print(custom_training_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "9a4cb3ccdcf2" + }, + "source": [ + "### [projects.jobs.get](https://cloud.google.com/ai-platform/training/docs/reference/rest/v1/projects.jobs/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "542ab0d537ac" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "78d6ac966b97" + }, + "outputs": [], + "source": [ + "request = client.projects().jobs().get(name=custom_training_id)\n", + "\n", + "result = request.execute()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "8a321239ce25" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "eb937a901fcf" + }, + "outputs": [], + "source": [ + "print(json.dumps(result, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"jobId\": \"custom_job_SKL20210302140139\",\n", + " \"trainingInput\": {\n", + " \"packageUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210302140139/census.tar.gz\"\n", + " ],\n", + " \"pythonModule\": \"trainer.task\",\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210302140139/custom_job_SKL20210302140139\"\n", + " ],\n", + " \"region\": \"us-central1\",\n", + " \"runtimeVersion\": \"2.4\",\n", + " \"pythonVersion\": \"3.7\"\n", + " },\n", + " \"createTime\": \"2021-03-02T14:09:33Z\",\n", + " \"state\": \"PREPARING\",\n", + " \"trainingOutput\": {},\n", + " \"etag\": \"/X2Bt4OWbWU=\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = client.projects().jobs().get(name=custom_training_id).execute()\n", + "\n", + " if response[\"state\"] != \"SUCCEEDED\":\n", + " print(\"Training job has not completed:\", response[\"state\"])\n", + " if response[\"state\"] == \"FAILED\":\n", + " break\n", + " else:\n", + " break\n", + " time.sleep(60)\n", + "\n", + "# model artifact output directory on Google Cloud Storage\n", + "model_artifact_dir = response[\"trainingInput\"][\"args\"][0].split(\"=\")[-1]\n", + "print(\"artifact location \" + model_artifact_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0a9277418cf4" + }, + "source": [ + "## Deploy the model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "50c25f9a03e6" + }, + "source": [ + "### [projects.models.create](https://cloud.google.com/ai-platform/prediction/docs/reference/rest/v1/projects.models/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "d00c48a15512" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "2dd3b73dcc8b" + }, + "outputs": [], + "source": [ + "body = {\"name\": \"custom_job_SKL\" + TIMESTAMP}\n", + "\n", + "request = client.projects().models().create(parent=\"projects/\" + PROJECT_ID)\n", + "request.body = json.loads(json.dumps(body, indent=2))\n", + "\n", + "print(json.dumps(json.loads(request.to_json()), indent=2))\n", + "\n", + "request = client.projects().models().create(parent=\"projects/\" + PROJECT_ID, body=body)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"uri\": \"https://ml.googleapis.com/v1/projects/migration-ucaip-training/models?alt=json\",\n", + " \"method\": \"POST\",\n", + " \"body\": {\n", + " \"name\": \"custom_job_SKL20210302140139\"\n", + " },\n", + " \"headers\": {\n", + " \"accept\": \"application/json\",\n", + " \"accept-encoding\": \"gzip, deflate\",\n", + " \"user-agent\": \"(gzip)\",\n", + " \"x-goog-api-client\": \"gdcl/1.12.8 gl-python/3.7.8\"\n", + " },\n", + " \"methodId\": \"ml.projects.models.create\",\n", + " \"resumable\": null,\n", + " \"response_callbacks\": [],\n", + " \"_in_error_state\": false,\n", + " \"body_size\": 0,\n", + " \"resumable_uri\": null,\n", + " \"resumable_progress\": 0\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "41361264fb39" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "fa1ffb46eeed" + }, + "outputs": [], + "source": [ + "result = request.execute()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0010c38333b2" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "1a07cedac864" + }, + "outputs": [], + "source": [ + "print(json.dumps(result, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/migration-ucaip-training/models/custom_job_SKL20210302140139\",\n", + " \"regions\": [\n", + " \"us-central1\"\n", + " ],\n", + " \"etag\": \"Lmd8u9MSSIA=\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cd5ebb78dedb" + }, + "outputs": [], + "source": [ + "model_id = result[\"name\"]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "78a109a6e197" + }, + "source": [ + "### [projects.models.versions.create](https://cloud.google.com/ai-platform/prediction/docs/reference/rest/v1/projects.models.versions/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "b425dd724634" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "f656b4b49c49" + }, + "outputs": [], + "source": [ + "version = {\n", + " \"name\": \"custom_job_SKL\" + TIMESTAMP,\n", + " \"deploymentUri\": model_artifact_dir,\n", + " \"runtimeVersion\": \"2.1\",\n", + " \"framework\": \"SCIKIT_LEARN\",\n", + " \"pythonVersion\": \"3.7\",\n", + " \"machineType\": \"mls1-c1-m2\",\n", + "}\n", + "\n", + "request = client.projects().models().versions().create(parent=model_id)\n", + "request.body = version\n", + "\n", + "print(json.dumps(json.loads(request.to_json()), indent=2))\n", + "\n", + "request = client.projects().models().versions().create(parent=model_id, body=version)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"uri\": \"https://ml.googleapis.com/v1/projects/migration-ucaip-training/models/custom_job_SKL20210302140139/versions?alt=json\",\n", + " \"method\": \"POST\",\n", + " \"body\": {\n", + " \"name\": \"custom_job_SKL20210302140139\",\n", + " \"deploymentUri\": \"gs://migration-ucaip-trainingaip-20210302140139/custom_job_SKL20210302140139\",\n", + " \"runtimeVersion\": \"2.1\",\n", + " \"framework\": \"SCIKIT_LEARN\",\n", + " \"pythonVersion\": \"3.7\",\n", + " \"machineType\": \"mls1-c1-m2\"\n", + " },\n", + " \"headers\": {\n", + " \"accept\": \"application/json\",\n", + " \"accept-encoding\": \"gzip, deflate\",\n", + " \"user-agent\": \"(gzip)\",\n", + " \"x-goog-api-client\": \"gdcl/1.12.8 gl-python/3.7.8\"\n", + " },\n", + " \"methodId\": \"ml.projects.models.versions.create\",\n", + " \"resumable\": null,\n", + " \"response_callbacks\": [],\n", + " \"_in_error_state\": false,\n", + " \"body_size\": 0,\n", + " \"resumable_uri\": null,\n", + " \"resumable_progress\": 0\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bcc87289c1ec" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "f03d4ac1e2e1" + }, + "outputs": [], + "source": [ + "result = request.execute()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "94c9c1ca71d6" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "2734b8b9a16c" + }, + "outputs": [], + "source": [ + "print(json.dumps(result, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/migration-ucaip-training/operations/create_custom_job_SKL20210302140139_custom_job_SKL20210302140139-1614695138432\",\n", + " \"metadata\": {\n", + " \"@type\": \"type.googleapis.com/google.cloud.ml.v1.OperationMetadata\",\n", + " \"createTime\": \"2021-03-02T14:25:38Z\",\n", + " \"operationType\": \"CREATE_VERSION\",\n", + " \"modelName\": \"projects/migration-ucaip-training/models/custom_job_SKL20210302140139\",\n", + " \"version\": {\n", + " \"name\": \"projects/migration-ucaip-training/models/custom_job_SKL20210302140139/versions/custom_job_SKL20210302140139\",\n", + " \"deploymentUri\": \"gs://migration-ucaip-trainingaip-20210302140139/custom_job_SKL20210302140139\",\n", + " \"createTime\": \"2021-03-02T14:25:38Z\",\n", + " \"runtimeVersion\": \"2.1\",\n", + " \"etag\": \"ilPQVTiR+IM=\",\n", + " \"framework\": \"SCIKIT_LEARN\",\n", + " \"machineType\": \"mls1-c1-m2\",\n", + " \"pythonVersion\": \"3.7\"\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "996fc9f9148b" + }, + "outputs": [], + "source": [ + "# The full unique ID for the model version\n", + "model_version_name = result[\"metadata\"][\"version\"][\"name\"]\n", + "\n", + "print(model_version_name)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "be0930dcaf17" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = (\n", + " client.projects().models().versions().get(name=model_version_name).execute()\n", + " )\n", + " if response[\"state\"] == \"READY\":\n", + " print(\"Model version created.\")\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_predictions:migration" + }, + "source": [ + "## Make batch predictions\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "b934f438b6d7" + }, + "source": [ + "Batch prediction only supports Tensorflow. FRAMEWORK_SCIKIT_LEARN is not currently available." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "e5a08f8923cf" + }, + "source": [ + "## Make online predictions" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "21d33e9106e2" + }, + "source": [ + "### Prepare data item for online prediction" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bd5ebd856f93" + }, + "outputs": [], + "source": [ + "INSTANCES = [\n", + " [\n", + " 25,\n", + " \"Private\",\n", + " 226802,\n", + " \"11th\",\n", + " 7,\n", + " \"Never-married\",\n", + " \"Machine-op-inspct\",\n", + " \"Own-child\",\n", + " \"Black\",\n", + " \"Male\",\n", + " 0,\n", + " 0,\n", + " 40,\n", + " \"United-States\",\n", + " ],\n", + " [\n", + " 38,\n", + " \"Private\",\n", + " 89814,\n", + " \"HS-grad\",\n", + " 9,\n", + " \"Married-civ-spouse\",\n", + " \"Farming-fishing\",\n", + " \"Husband\",\n", + " \"White\",\n", + " \"Male\",\n", + " 0,\n", + " 0,\n", + " 50,\n", + " \"United-States\",\n", + " ],\n", + " [\n", + " 28,\n", + " \"Local-gov\",\n", + " 336951,\n", + " \"Assoc-acdm\",\n", + " 12,\n", + " \"Married-civ-spouse\",\n", + " \"Protective-serv\",\n", + " \"Husband\",\n", + " \"White\",\n", + " \"Male\",\n", + " 0,\n", + " 0,\n", + " 40,\n", + " \"United-States\",\n", + " ],\n", + " [\n", + " 44,\n", + " \"Private\",\n", + " 160323,\n", + " \"Some-college\",\n", + " 10,\n", + " \"Married-civ-spouse\",\n", + " \"Machine-op-inspct\",\n", + " \"Husband\",\n", + " \"Black\",\n", + " \"Male\",\n", + " 7688,\n", + " 0,\n", + " 40,\n", + " \"United-States\",\n", + " ],\n", + " [\n", + " 18,\n", + " \"?\",\n", + " 103497,\n", + " \"Some-college\",\n", + " 10,\n", + " \"Never-married\",\n", + " \"?\",\n", + " \"Own-child\",\n", + " \"White\",\n", + " \"Female\",\n", + " 0,\n", + " 0,\n", + " 30,\n", + " \"United-States\",\n", + " ],\n", + " [\n", + " 34,\n", + " \"Private\",\n", + " 198693,\n", + " \"10th\",\n", + " 6,\n", + " \"Never-married\",\n", + " \"Other-service\",\n", + " \"Not-in-family\",\n", + " \"White\",\n", + " \"Male\",\n", + " 0,\n", + " 0,\n", + " 30,\n", + " \"United-States\",\n", + " ],\n", + " [\n", + " 29,\n", + " \"?\",\n", + " 227026,\n", + " \"HS-grad\",\n", + " 9,\n", + " \"Never-married\",\n", + " \"?\",\n", + " \"Unmarried\",\n", + " \"Black\",\n", + " \"Male\",\n", + " 0,\n", + " 0,\n", + " 40,\n", + " \"United-States\",\n", + " ],\n", + " [\n", + " 63,\n", + " \"Self-emp-not-inc\",\n", + " 104626,\n", + " \"Prof-school\",\n", + " 15,\n", + " \"Married-civ-spouse\",\n", + " \"Prof-specialty\",\n", + " \"Husband\",\n", + " \"White\",\n", + " \"Male\",\n", + " 3103,\n", + " 0,\n", + " 32,\n", + " \"United-States\",\n", + " ],\n", + " [\n", + " 24,\n", + " \"Private\",\n", + " 369667,\n", + " \"Some-college\",\n", + " 10,\n", + " \"Never-married\",\n", + " \"Other-service\",\n", + " \"Unmarried\",\n", + " \"White\",\n", + " \"Female\",\n", + " 0,\n", + " 0,\n", + " 40,\n", + " \"United-States\",\n", + " ],\n", + " [\n", + " 55,\n", + " \"Private\",\n", + " 104996,\n", + " \"7th-8th\",\n", + " 4,\n", + " \"Married-civ-spouse\",\n", + " \"Craft-repair\",\n", + " \"Husband\",\n", + " \"White\",\n", + " \"Male\",\n", + " 0,\n", + " 0,\n", + " 10,\n", + " \"United-States\",\n", + " ],\n", + "]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "6b3eb0485e07" + }, + "source": [ + "### [projects.predict](https://cloud.google.com/ai-platform/prediction/docs/reference/rest/v1/projects/predict)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "78235f0af030" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "1d080e69fe0d" + }, + "outputs": [], + "source": [ + "request = client.projects().predict(name=model_version_name)\n", + "request.body = json.loads(json.dumps({\"instances\": INSTANCES}, indent=2))\n", + "\n", + "print(json.dumps(json.loads(request.to_json()), indent=2))\n", + "\n", + "request = client.projects().predict(\n", + " name=model_version_name, body={\"instances\": INSTANCES}\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"uri\": \"https://ml.googleapis.com/v1/projects/migration-ucaip-training/models/custom_job_SKL20210302140139/versions/custom_job_SKL20210302140139:predict?alt=json\",\n", + " \"method\": \"POST\",\n", + " \"body\": {\n", + " \"instances\": [\n", + " [\n", + " 25,\n", + " \"Private\",\n", + " 226802,\n", + " \"11th\",\n", + " 7,\n", + " \"Never-married\",\n", + " \"Machine-op-inspct\",\n", + " \"Own-child\",\n", + " \"Black\",\n", + " \"Male\",\n", + " 0,\n", + " 0,\n", + " 40,\n", + " \"United-States\"\n", + " ],\n", + " [\n", + " 38,\n", + " \"Private\",\n", + " 89814,\n", + " \"HS-grad\",\n", + " 9,\n", + " \"Married-civ-spouse\",\n", + " \"Farming-fishing\",\n", + " \"Husband\",\n", + " \"White\",\n", + " \"Male\",\n", + " 0,\n", + " 0,\n", + " 50,\n", + " \"United-States\"\n", + " ],\n", + " [\n", + " 28,\n", + " \"Local-gov\",\n", + " 336951,\n", + " \"Assoc-acdm\",\n", + " 12,\n", + " \"Married-civ-spouse\",\n", + " \"Protective-serv\",\n", + " \"Husband\",\n", + " \"White\",\n", + " \"Male\",\n", + " 0,\n", + " 0,\n", + " 40,\n", + " \"United-States\"\n", + " ],\n", + " [\n", + " 44,\n", + " \"Private\",\n", + " 160323,\n", + " \"Some-college\",\n", + " 10,\n", + " \"Married-civ-spouse\",\n", + " \"Machine-op-inspct\",\n", + " \"Husband\",\n", + " \"Black\",\n", + " \"Male\",\n", + " 7688,\n", + " 0,\n", + " 40,\n", + " \"United-States\"\n", + " ],\n", + " [\n", + " 18,\n", + " \"?\",\n", + " 103497,\n", + " \"Some-college\",\n", + " 10,\n", + " \"Never-married\",\n", + " \"?\",\n", + " \"Own-child\",\n", + " \"White\",\n", + " \"Female\",\n", + " 0,\n", + " 0,\n", + " 30,\n", + " \"United-States\"\n", + " ],\n", + " [\n", + " 34,\n", + " \"Private\",\n", + " 198693,\n", + " \"10th\",\n", + " 6,\n", + " \"Never-married\",\n", + " \"Other-service\",\n", + " \"Not-in-family\",\n", + " \"White\",\n", + " \"Male\",\n", + " 0,\n", + " 0,\n", + " 30,\n", + " \"United-States\"\n", + " ],\n", + " [\n", + " 29,\n", + " \"?\",\n", + " 227026,\n", + " \"HS-grad\",\n", + " 9,\n", + " \"Never-married\",\n", + " \"?\",\n", + " \"Unmarried\",\n", + " \"Black\",\n", + " \"Male\",\n", + " 0,\n", + " 0,\n", + " 40,\n", + " \"United-States\"\n", + " ],\n", + " [\n", + " 63,\n", + " \"Self-emp-not-inc\",\n", + " 104626,\n", + " \"Prof-school\",\n", + " 15,\n", + " \"Married-civ-spouse\",\n", + " \"Prof-specialty\",\n", + " \"Husband\",\n", + " \"White\",\n", + " \"Male\",\n", + " 3103,\n", + " 0,\n", + " 32,\n", + " \"United-States\"\n", + " ],\n", + " [\n", + " 24,\n", + " \"Private\",\n", + " 369667,\n", + " \"Some-college\",\n", + " 10,\n", + " \"Never-married\",\n", + " \"Other-service\",\n", + " \"Unmarried\",\n", + " \"White\",\n", + " \"Female\",\n", + " 0,\n", + " 0,\n", + " 40,\n", + " \"United-States\"\n", + " ],\n", + " [\n", + " 55,\n", + " \"Private\",\n", + " 104996,\n", + " \"7th-8th\",\n", + " 4,\n", + " \"Married-civ-spouse\",\n", + " \"Craft-repair\",\n", + " \"Husband\",\n", + " \"White\",\n", + " \"Male\",\n", + " 0,\n", + " 0,\n", + " 10,\n", + " \"United-States\"\n", + " ]\n", + " ]\n", + " },\n", + " \"headers\": {\n", + " \"accept\": \"application/json\",\n", + " \"accept-encoding\": \"gzip, deflate\",\n", + " \"user-agent\": \"(gzip)\",\n", + " \"x-goog-api-client\": \"gdcl/1.12.8 gl-python/3.7.8\"\n", + " },\n", + " \"methodId\": \"ml.projects.predict\",\n", + " \"resumable\": null,\n", + " \"response_callbacks\": [],\n", + " \"_in_error_state\": false,\n", + " \"body_size\": 0,\n", + " \"resumable_uri\": null,\n", + " \"resumable_progress\": 0\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aae42336f6d8" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "792b1168b16c" + }, + "outputs": [], + "source": [ + "result = request.execute()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "f331f79f6345" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ecc8a33f5f99" + }, + "outputs": [], + "source": [ + "print(json.dumps(result, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"predictions\": [\n", + " false,\n", + " false,\n", + " false,\n", + " false,\n", + " false,\n", + " false,\n", + " false,\n", + " false,\n", + " false,\n", + " false\n", + " ]\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "d8362aa15a50" + }, + "source": [ + "### [projects.models.versions.delete](https://cloud.google.com/ai-platform/prediction/docs/reference/rest/v1/projects.models.versions/delete)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "72d35ff235e6" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "08a979e387d5" + }, + "outputs": [], + "source": [ + "request = client.projects().models().versions().delete(name=model_version_name)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "4c4f0aa6241a" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "94b3a55f1cc6" + }, + "outputs": [], + "source": [ + "response = request.execute()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "response:migration" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,request" + }, + "outputs": [], + "source": [ + "print(json.dumps(response, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/migration-ucaip-training/operations/delete_custom_job_SKL20210302140139_custom_job_SKL20210302140139-1614695211809\",\n", + " \"metadata\": {\n", + " \"@type\": \"type.googleapis.com/google.cloud.ml.v1.OperationMetadata\",\n", + " \"createTime\": \"2021-03-02T14:26:51Z\",\n", + " \"operationType\": \"DELETE_VERSION\",\n", + " \"modelName\": \"projects/migration-ucaip-training/models/custom_job_SKL20210302140139\",\n", + " \"version\": {\n", + " \"name\": \"projects/migration-ucaip-training/models/custom_job_SKL20210302140139/versions/custom_job_SKL20210302140139\",\n", + " \"deploymentUri\": \"gs://migration-ucaip-trainingaip-20210302140139/custom_job_SKL20210302140139\",\n", + " \"createTime\": \"2021-03-02T14:25:38Z\",\n", + " \"runtimeVersion\": \"2.1\",\n", + " \"state\": \"READY\",\n", + " \"etag\": \"5R4YqeqWMk8=\",\n", + " \"framework\": \"SCIKIT_LEARN\",\n", + " \"machineType\": \"mls1-c1-m2\",\n", + " \"pythonVersion\": \"3.7\"\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup:migration,new" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup:migration,new" + }, + "outputs": [], + "source": [ + "delete_model = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model:\n", + " client.projects().models().delete(name=model_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r gs://$BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "UJ10 legacy Custom Training Prebuilt Container SKLearn.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/migration/UJ10 unified Custom Training Prebuilt Container SKLearn.ipynb b/notebooks/community/migration/UJ10 unified Custom Training Prebuilt Container SKLearn.ipynb new file mode 100644 index 000000000..d9c29a7e4 --- /dev/null +++ b/notebooks/community/migration/UJ10 unified Custom Training Prebuilt Container SKLearn.ipynb @@ -0,0 +1,2649 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title:migration,new" + }, + "source": [ + "# Vertex SDK: Train and deploy an SKLearn model with pre-built containers (formerly hosted runtimes)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest (preview) version of Vertex SDK.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-aiplatform --user" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the Google *cloud-storage* library as well.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-storage" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the Kernel\n", + "\n", + "Once you've installed the Vertex SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "if not os.getenv(\"AUTORUN\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU run-time\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your GCP project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a GCP project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebooks.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex AI. We recommend when possible, to choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You cannot use a Multi-Regional Storage bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see [Region support for Vertex AI services](https://cloud.google.com/vertex-ai/docs/general/locations)\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your GCP account\n", + "\n", + "**If you are using Google Cloud Notebooks**, your environment is already\n", + "authenticated. Skip this step.\n", + "\n", + "*Note: If you are on an Vertex notebook and run the cell, the cell knows to skip executing the authentication steps.*\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your Google Cloud account. This provides access\n", + "# to your Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Vertex, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this tutorial in a notebook locally, replace the string\n", + " # below with the path to your service account key and run this cell to\n", + " # authenticate your Google Cloud account.\n", + " else:\n", + " %env GOOGLE_APPLICATION_CREDENTIALS your_path_to_credentials.json\n", + "\n", + " # Log in to your account on Google Cloud\n", + " ! gcloud auth login" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n", + " BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip" + }, + "source": [ + "#### Import Vertex SDK\n", + "\n", + "Import the Vertex SDK into our Python environment.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex AI constants\n", + "\n", + "Setup up the following constants for Vertex AI:\n", + "\n", + "- `API_ENDPOINT`: The Vertex AI API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex AI location root path for dataset, model and endpoint resources.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API Endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex AI location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients" + }, + "source": [ + "## Clients\n", + "\n", + "The Vertex SDK works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the server (Vertex).\n", + "\n", + "You will use several clients in this tutorial, so set them all up upfront.\n", + "\n", + "- Model Service for managed models.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "clients[\"job\"] = create_job_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0ce08bfdc2d0" + }, + "source": [ + "## Prepare a trainer script" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "1e930837e6a2" + }, + "source": [ + "### Package assembly" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "4f1f6164f0fa" + }, + "outputs": [], + "source": [ + "# Make folder for python training script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\\n", + "tag_build =\\n\\\n", + "tag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\\n", + "setuptools.setup(\\n\\\n", + " install_requires=[\\n\\\n", + " ],\\n\\\n", + " packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\\n", + "Name: Custom Census Income\\n\\\n", + "Version: 0.0.0\\n\\\n", + "Summary: Demonstration training script\\n\\\n", + "Home-page: www.google.com\\n\\\n", + "Author: Google\\n\\\n", + "Author-email: aferlitsch@google.com\\n\\\n", + "License: Public\\n\\\n", + "Description: Demo\\n\\\n", + "Platform: Vertex AI\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "40d7d11f7de7" + }, + "source": [ + "### Task.py contents" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "96d9bd80ddda" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Single Instance Training for Census Income\n", + "\n", + "from sklearn.ensemble import RandomForestClassifier\n", + "import joblib\n", + "from sklearn.feature_selection import SelectKBest\n", + "from sklearn.pipeline import FeatureUnion\n", + "from sklearn.pipeline import Pipeline\n", + "from sklearn.preprocessing import LabelBinarizer\n", + "import datetime\n", + "import pandas as pd\n", + "\n", + "from google.cloud import storage\n", + "\n", + "import numpy as np\n", + "import argparse\n", + "import os\n", + "import sys\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv('AIP_MODEL_DIR'), type=str, help='Model dir.')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "\n", + "# Public bucket holding the census data\n", + "bucket = storage.Client().bucket('cloud-samples-data')\n", + "\n", + "# Path to the data inside the public bucket\n", + "blob = bucket.blob('ai-platform/sklearn/census_data/adult.data')\n", + "# Download the data\n", + "blob.download_to_filename('adult.data')\n", + "\n", + "# Define the format of your input data including unused columns (These are the columns from the census data files)\n", + "COLUMNS = (\n", + " 'age',\n", + " 'workclass',\n", + " 'fnlwgt',\n", + " 'education',\n", + " 'education-num',\n", + " 'marital-status',\n", + " 'occupation',\n", + " 'relationship',\n", + " 'race',\n", + " 'sex',\n", + " 'capital-gain',\n", + " 'capital-loss',\n", + " 'hours-per-week',\n", + " 'native-country',\n", + " 'income-level'\n", + ")\n", + "\n", + "# Categorical columns are columns that need to be turned into a numerical value to be used by scikit-learn\n", + "CATEGORICAL_COLUMNS = (\n", + " 'workclass',\n", + " 'education',\n", + " 'marital-status',\n", + " 'occupation',\n", + " 'relationship',\n", + " 'race',\n", + " 'sex',\n", + " 'native-country'\n", + ")\n", + "\n", + "\n", + "# Load the training census dataset\n", + "with open('./adult.data', 'r') as train_data:\n", + " raw_training_data = pd.read_csv(train_data, header=None, names=COLUMNS)\n", + "\n", + "# Remove the column we are trying to predict ('income-level') from our features list\n", + "# Convert the Dataframe to a lists of lists\n", + "train_features = raw_training_data.drop('income-level', axis=1).values.tolist()\n", + "# Create our training labels list, convert the Dataframe to a lists of lists\n", + "train_labels = (raw_training_data['income-level'] == ' >50K').values.tolist()\n", + "\n", + "# Since the census data set has categorical features, we need to convert\n", + "# them to numerical values. We'll use a list of pipelines to convert each\n", + "# categorical column and then use FeatureUnion to combine them before calling\n", + "# the RandomForestClassifier.\n", + "categorical_pipelines = []\n", + "\n", + "# Each categorical column needs to be extracted individually and converted to a numerical value.\n", + "# To do this, each categorical column will use a pipeline that extracts one feature column via\n", + "# SelectKBest(k=1) and a LabelBinarizer() to convert the categorical value to a numerical one.\n", + "# A scores array (created below) will select and extract the feature column. The scores array is\n", + "# created by iterating over the COLUMNS and checking if it is a CATEGORICAL_COLUMN.\n", + "for i, col in enumerate(COLUMNS[:-1]):\n", + " if col in CATEGORICAL_COLUMNS:\n", + " # Create a scores array to get the individual categorical column.\n", + " # Example:\n", + " # data = [39, 'State-gov', 77516, 'Bachelors', 13, 'Never-married', 'Adm-clerical',\n", + " # 'Not-in-family', 'White', 'Male', 2174, 0, 40, 'United-States']\n", + " # scores = [0, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0]\n", + " #\n", + " # Returns: [['State-gov']]\n", + " # Build the scores array.\n", + " scores = [0] * len(COLUMNS[:-1])\n", + " # This column is the categorical column we want to extract.\n", + " scores[i] = 1\n", + " skb = SelectKBest(k=1)\n", + " skb.scores_ = scores\n", + " # Convert the categorical column to a numerical value\n", + " lbn = LabelBinarizer()\n", + " r = skb.transform(train_features)\n", + " lbn.fit(r)\n", + " # Create the pipeline to extract the categorical feature\n", + " categorical_pipelines.append(\n", + " ('categorical-{}'.format(i), Pipeline([\n", + " ('SKB-{}'.format(i), skb),\n", + " ('LBN-{}'.format(i), lbn)])))\n", + " \n", + "# Create pipeline to extract the numerical features\n", + "skb = SelectKBest(k=6)\n", + "# From COLUMNS use the features that are numerical\n", + "skb.scores_ = [1, 0, 1, 0, 1, 0, 0, 0, 0, 0, 1, 1, 1, 0]\n", + "categorical_pipelines.append(('numerical', skb))\n", + "\n", + "# Combine all the features using FeatureUnion\n", + "preprocess = FeatureUnion(categorical_pipelines)\n", + "\n", + "# Create the classifier\n", + "classifier = RandomForestClassifier()\n", + "\n", + "# Transform the features and fit them to the classifier\n", + "classifier.fit(preprocess.transform(train_features), train_labels)\n", + "\n", + "# Create the overall model as a single pipeline\n", + "pipeline = Pipeline([\n", + " ('union', preprocess),\n", + " ('classifier', classifier)\n", + "])\n", + "\n", + "# Split path into bucket and subdirectory\n", + "bucket = args.model_dir.split('/')[2]\n", + "subdir = args.model_dir.split('/')[-1]\n", + "\n", + "# Write model to a local file\n", + "joblib.dump(pipeline, 'model.joblib')\n", + "\n", + "# Upload the model to GCS\n", + "bucket = storage.Client().bucket(bucket)\n", + "blob = bucket.blob(subdir + '/model.joblib')\n", + "blob.upload_from_filename('model.joblib')\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "3022bce50fbf" + }, + "source": [ + "### Store training script on your Cloud Storage bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "13409553e08b" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz gs://$BUCKET_NAME/census.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_create_and_deploy_model:migration" + }, + "source": [ + "## Train a model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0oqIBOSnJjkW" + }, + "source": [ + "### [projects.locations.customJobs.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.trainingPipelines/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "e110f8131d32" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "3c7288151426" + }, + "outputs": [], + "source": [ + "TRAIN_IMAGE = \"gcr.io/cloud-aiplatform/training/scikit-learn-cpu.0-23:latest\"\n", + "\n", + "JOB_NAME = \"custom_job_SKL\" + TIMESTAMP\n", + "\n", + "WORKER_POOL_SPEC = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": {\"machine_type\": \"n1-standard-4\"},\n", + " \"python_package_spec\": {\n", + " \"executor_image_uri\": TRAIN_IMAGE,\n", + " \"package_uris\": [\"gs://\" + BUCKET_NAME + \"/census.tar.gz\"],\n", + " \"python_module\": \"trainer.task\",\n", + " \"args\": [\"--model-dir=\" + \"gs://{}/{}\".format(BUCKET_NAME, JOB_NAME)],\n", + " },\n", + " }\n", + "]\n", + "\n", + "training_job = aip.CustomJob(\n", + " display_name=JOB_NAME, job_spec={\"worker_pool_specs\": WORKER_POOL_SPEC}\n", + ")\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateCustomJobRequest(parent=PARENT, custom_job=training_job).__dict__[\n", + " \"_pb\"\n", + " ]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"customJob\": {\n", + " \"displayName\": \"custom_job_SKL20210323185534\",\n", + " \"jobSpec\": {\n", + " \"workerPoolSpecs\": [\n", + " {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-4\"\n", + " },\n", + " \"replicaCount\": \"1\",\n", + " \"pythonPackageSpec\": {\n", + " \"executorImageUri\": \"gcr.io/cloud-aiplatform/training/scikit-learn-cpu.0-23:latest\",\n", + " \"packageUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210323185534/census.tar.gz\"\n", + " ],\n", + " \"pythonModule\": \"trainer.task\",\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210323185534/custom_job_SKL20210323185534\"\n", + " ]\n", + " }\n", + " }\n", + " ]\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "1fcd4e82a52b" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "fbe59127c6f6" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].create_custom_job(parent=PARENT, custom_job=training_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "3dffa1c62454" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "4086d1e46b00" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/customJobs/3216493723709865984\",\n", + " \"displayName\": \"custom_job_SKL20210323185534\",\n", + " \"jobSpec\": {\n", + " \"workerPoolSpecs\": [\n", + " {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-4\"\n", + " },\n", + " \"replicaCount\": \"1\",\n", + " \"diskSpec\": {\n", + " \"bootDiskType\": \"pd-ssd\",\n", + " \"bootDiskSizeGb\": 100\n", + " },\n", + " \"pythonPackageSpec\": {\n", + " \"executorImageUri\": \"gcr.io/cloud-aiplatform/training/scikit-learn-cpu.0-23:latest\",\n", + " \"packageUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210323185534/census.tar.gz\"\n", + " ],\n", + " \"pythonModule\": \"trainer.task\",\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210323185534/custom_job_SKL20210323185534\"\n", + " ]\n", + " }\n", + " }\n", + " ]\n", + " },\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"createTime\": \"2021-03-23T18:55:41.688375Z\",\n", + " \"updateTime\": \"2021-03-23T18:55:41.688375Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "training_pipeline_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom training job\n", + "custom_training_id = request.name\n", + "# The short numeric ID for the custom training job\n", + "custom_training_short_id = custom_training_id.split(\"/\")[-1]\n", + "\n", + "print(custom_training_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0oqIBOSnJjkW" + }, + "source": [ + "### [projects.locations.customJobs.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.trainingPipelines/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dd8e5e3427d5" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "734b3788cff1" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].get_custom_job(name=custom_training_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "f145335bc684" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "30eef648d2ec" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/customJobs/3216493723709865984\",\n", + " \"displayName\": \"custom_job_SKL20210323185534\",\n", + " \"jobSpec\": {\n", + " \"workerPoolSpecs\": [\n", + " {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-4\"\n", + " },\n", + " \"replicaCount\": \"1\",\n", + " \"diskSpec\": {\n", + " \"bootDiskType\": \"pd-ssd\",\n", + " \"bootDiskSizeGb\": 100\n", + " },\n", + " \"pythonPackageSpec\": {\n", + " \"executorImageUri\": \"gcr.io/cloud-aiplatform/training/scikit-learn-cpu.0-23:latest\",\n", + " \"packageUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210323185534/census.tar.gz\"\n", + " ],\n", + " \"pythonModule\": \"trainer.task\",\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210323185534/custom_job_SKL20210323185534\"\n", + " ]\n", + " }\n", + " }\n", + " ]\n", + " },\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"createTime\": \"2021-03-23T18:55:41.688375Z\",\n", + " \"updateTime\": \"2021-03-23T18:55:41.688375Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = clients[\"job\"].get_custom_job(name=custom_training_id)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " break\n", + " else:\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "# model artifact output directory on Google Cloud Storage\n", + "model_artifact_dir = (\n", + " response.job_spec.worker_pool_specs[0].python_package_spec.args[0].split(\"=\")[-1]\n", + ")\n", + "print(\"artifact location \" + model_artifact_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "576a7e7ce36c" + }, + "source": [ + "## Deploy the model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "COwVZtxhJjkW" + }, + "source": [ + "### [projects.locations.models.upload](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.models/upload)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "fed9fd1f70cf" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aedf59150ec9" + }, + "outputs": [], + "source": [ + "DEPLOY_IMAGE = \"gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-23:latest\"\n", + "\n", + "model = {\n", + " \"display_name\": \"custom_job_SKL\" + TIMESTAMP,\n", + " \"artifact_uri\": model_artifact_dir,\n", + " \"container_spec\": {\"image_uri\": DEPLOY_IMAGE, \"ports\": [{\"container_port\": 8080}]},\n", + "}\n", + "\n", + "print(MessageToJson(aip.UploadModelRequest(parent=PARENT, model=model).__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"model\": {\n", + " \"displayName\": \"custom_job_SKL20210323185534\",\n", + " \"containerSpec\": {\n", + " \"imageUri\": \"gcr.io/cloud-aiplatform/prediction/sklearn-cpu.0-23:latest\",\n", + " \"ports\": [\n", + " {\n", + " \"containerPort\": 8080\n", + " }\n", + " ]\n", + " },\n", + " \"artifactUri\": \"gs://migration-ucaip-trainingaip-20210323185534/custom_job_SKL20210323185534\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "8641495cc6f9" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "5a967bc5db1b" + }, + "outputs": [], + "source": [ + "request = clients[\"model\"].upload_model(parent=PARENT, model=model)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "2118f3c18d0f" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "1de1269f8216" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/5984808915752189952\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "0b2a551124e1" + }, + "outputs": [], + "source": [ + "model_id = result.model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_predictions:migration" + }, + "source": [ + "## Make batch predictions\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_prediction_file:migration,new" + }, + "source": [ + "### Make a batch prediction file\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_items:automl,icn,csv" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "import tensorflow as tf\n", + "\n", + "INSTANCES = [\n", + " [\n", + " 25,\n", + " \"Private\",\n", + " 226802,\n", + " \"11th\",\n", + " 7,\n", + " \"Never-married\",\n", + " \"Machine-op-inspct\",\n", + " \"Own-child\",\n", + " \"Black\",\n", + " \"Male\",\n", + " 0,\n", + " 0,\n", + " 40,\n", + " \"United-States\",\n", + " ],\n", + " [\n", + " 38,\n", + " \"Private\",\n", + " 89814,\n", + " \"HS-grad\",\n", + " 9,\n", + " \"Married-civ-spouse\",\n", + " \"Farming-fishing\",\n", + " \"Husband\",\n", + " \"White\",\n", + " \"Male\",\n", + " 0,\n", + " 0,\n", + " 50,\n", + " \"United-States\",\n", + " ],\n", + " [\n", + " 28,\n", + " \"Local-gov\",\n", + " 336951,\n", + " \"Assoc-acdm\",\n", + " 12,\n", + " \"Married-civ-spouse\",\n", + " \"Protective-serv\",\n", + " \"Husband\",\n", + " \"White\",\n", + " \"Male\",\n", + " 0,\n", + " 0,\n", + " 40,\n", + " \"United-States\",\n", + " ],\n", + " [\n", + " 44,\n", + " \"Private\",\n", + " 160323,\n", + " \"Some-college\",\n", + " 10,\n", + " \"Married-civ-spouse\",\n", + " \"Machine-op-inspct\",\n", + " \"Husband\",\n", + " \"Black\",\n", + " \"Male\",\n", + " 7688,\n", + " 0,\n", + " 40,\n", + " \"United-States\",\n", + " ],\n", + " [\n", + " 18,\n", + " \"?\",\n", + " 103497,\n", + " \"Some-college\",\n", + " 10,\n", + " \"Never-married\",\n", + " \"?\",\n", + " \"Own-child\",\n", + " \"White\",\n", + " \"Female\",\n", + " 0,\n", + " 0,\n", + " 30,\n", + " \"United-States\",\n", + " ],\n", + " [\n", + " 34,\n", + " \"Private\",\n", + " 198693,\n", + " \"10th\",\n", + " 6,\n", + " \"Never-married\",\n", + " \"Other-service\",\n", + " \"Not-in-family\",\n", + " \"White\",\n", + " \"Male\",\n", + " 0,\n", + " 0,\n", + " 30,\n", + " \"United-States\",\n", + " ],\n", + " [\n", + " 29,\n", + " \"?\",\n", + " 227026,\n", + " \"HS-grad\",\n", + " 9,\n", + " \"Never-married\",\n", + " \"?\",\n", + " \"Unmarried\",\n", + " \"Black\",\n", + " \"Male\",\n", + " 0,\n", + " 0,\n", + " 40,\n", + " \"United-States\",\n", + " ],\n", + " [\n", + " 63,\n", + " \"Self-emp-not-inc\",\n", + " 104626,\n", + " \"Prof-school\",\n", + " 15,\n", + " \"Married-civ-spouse\",\n", + " \"Prof-specialty\",\n", + " \"Husband\",\n", + " \"White\",\n", + " \"Male\",\n", + " 3103,\n", + " 0,\n", + " 32,\n", + " \"United-States\",\n", + " ],\n", + " [\n", + " 24,\n", + " \"Private\",\n", + " 369667,\n", + " \"Some-college\",\n", + " 10,\n", + " \"Never-married\",\n", + " \"Other-service\",\n", + " \"Unmarried\",\n", + " \"White\",\n", + " \"Female\",\n", + " 0,\n", + " 0,\n", + " 40,\n", + " \"United-States\",\n", + " ],\n", + " [\n", + " 55,\n", + " \"Private\",\n", + " 104996,\n", + " \"7th-8th\",\n", + " 4,\n", + " \"Married-civ-spouse\",\n", + " \"Craft-repair\",\n", + " \"Husband\",\n", + " \"White\",\n", + " \"Male\",\n", + " 0,\n", + " 0,\n", + " 10,\n", + " \"United-States\",\n", + " ],\n", + "]\n", + "\n", + "gcs_input_uri = \"gs://\" + BUCKET_NAME + \"/\" + \"test.jsonl\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " for i in INSTANCES:\n", + " f.write(json.dumps(i) + \"\\n\")\n", + "\n", + "! gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "[25, \"Private\", 226802, \"11th\", 7, \"Never-married\", \"Machine-op-inspct\", \"Own-child\", \"Black\", \"Male\", 0, 0, 40, \"United-States\"]\n", + "[38, \"Private\", 89814, \"HS-grad\", 9, \"Married-civ-spouse\", \"Farming-fishing\", \"Husband\", \"White\", \"Male\", 0, 0, 50, \"United-States\"]\n", + "[28, \"Local-gov\", 336951, \"Assoc-acdm\", 12, \"Married-civ-spouse\", \"Protective-serv\", \"Husband\", \"White\", \"Male\", 0, 0, 40, \"United-States\"]\n", + "[44, \"Private\", 160323, \"Some-college\", 10, \"Married-civ-spouse\", \"Machine-op-inspct\", \"Husband\", \"Black\", \"Male\", 7688, 0, 40, \"United-States\"]\n", + "[18, \"?\", 103497, \"Some-college\", 10, \"Never-married\", \"?\", \"Own-child\", \"White\", \"Female\", 0, 0, 30, \"United-States\"]\n", + "[34, \"Private\", 198693, \"10th\", 6, \"Never-married\", \"Other-service\", \"Not-in-family\", \"White\", \"Male\", 0, 0, 30, \"United-States\"]\n", + "[29, \"?\", 227026, \"HS-grad\", 9, \"Never-married\", \"?\", \"Unmarried\", \"Black\", \"Male\", 0, 0, 40, \"United-States\"]\n", + "[63, \"Self-emp-not-inc\", 104626, \"Prof-school\", 15, \"Married-civ-spouse\", \"Prof-specialty\", \"Husband\", \"White\", \"Male\", 3103, 0, 32, \"United-States\"]\n", + "[24, \"Private\", 369667, \"Some-college\", 10, \"Never-married\", \"Other-service\", \"Unmarried\", \"White\", \"Female\", 0, 0, 40, \"United-States\"]\n", + "[55, \"Private\", 104996, \"7th-8th\", 4, \"Married-civ-spouse\", \"Craft-repair\", \"Husband\", \"White\", \"Male\", 0, 0, 10, \"United-States\"]\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_create:migration,new" + }, + "source": [ + "### [projects.locations.batchPredictionJobs.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.batchPredictionJobs/create)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "request:migration" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_create:migration,new,request,icn" + }, + "outputs": [], + "source": [ + "model_parameters = Value(\n", + " struct_value=Struct(\n", + " fields={\n", + " \"confidence_threshold\": Value(number_value=0.5),\n", + " \"max_predictions\": Value(number_value=10000.0),\n", + " }\n", + " )\n", + ")\n", + "\n", + "batch_prediction_job = {\n", + " \"display_name\": \"custom_job_SKL\" + TIMESTAMP,\n", + " \"model\": model_id,\n", + " \"input_config\": {\n", + " \"instances_format\": \"jsonl\",\n", + " \"gcs_source\": {\"uris\": [gcs_input_uri]},\n", + " },\n", + " \"model_parameters\": model_parameters,\n", + " \"output_config\": {\n", + " \"predictions_format\": \"jsonl\",\n", + " \"gcs_destination\": {\n", + " \"output_uri_prefix\": \"gs://\" + f\"{BUCKET_NAME}/batch_output/\"\n", + " },\n", + " },\n", + " \"dedicated_resources\": {\n", + " \"machine_spec\": {\"machine_type\": \"n1-standard-2\"},\n", + " \"starting_replica_count\": 1,\n", + " \"max_replica_count\": 1,\n", + " },\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateBatchPredictionJobRequest(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_create:migration,new,request,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"batchPredictionJob\": {\n", + " \"displayName\": \"custom_job_SKL20210323185534\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/5984808915752189952\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210323185534/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"modelParameters\": {\n", + " \"confidence_threshold\": 0.5,\n", + " \"max_predictions\": 10000.0\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210323185534/batch_output/\"\n", + " }\n", + " },\n", + " \"dedicatedResources\": {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-2\"\n", + " },\n", + " \"startingReplicaCount\": 1,\n", + " \"maxReplicaCount\": 1\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "call:migration" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].create_batch_prediction_job(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "response:migration" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,request" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_create:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/batchPredictionJobs/2509428582212698112\",\n", + " \"displayName\": \"custom_job_SKL20210323185534\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/5984808915752189952\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210323185534/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"modelParameters\": {\n", + " \"max_predictions\": 10000.0,\n", + " \"confidence_threshold\": 0.5\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210323185534/batch_output/\"\n", + " }\n", + " },\n", + " \"dedicatedResources\": {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-2\"\n", + " },\n", + " \"startingReplicaCount\": 1,\n", + " \"maxReplicaCount\": 1\n", + " },\n", + " \"manualBatchTuningParameters\": {},\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"createTime\": \"2021-03-23T19:05:07.344290Z\",\n", + " \"updateTime\": \"2021-03-23T19:05:07.344290Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batch_job_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The fully qualified ID for the batch job\n", + "batch_job_id = request.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split(\"/\")[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_get:migration,new" + }, + "source": [ + "### [projects.locations.batchPredictionJobs.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.batchPredictionJobs/get)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "call:migration" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_get:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].get_batch_prediction_job(name=batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "response:migration" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,request" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_get:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/batchPredictionJobs/2509428582212698112\",\n", + " \"displayName\": \"custom_job_SKL20210323185534\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/5984808915752189952\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210323185534/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"modelParameters\": {\n", + " \"confidence_threshold\": 0.5,\n", + " \"max_predictions\": 10000.0\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210323185534/batch_output/\"\n", + " }\n", + " },\n", + " \"dedicatedResources\": {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-2\"\n", + " },\n", + " \"startingReplicaCount\": 1,\n", + " \"maxReplicaCount\": 1\n", + " },\n", + " \"manualBatchTuningParameters\": {},\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"createTime\": \"2021-03-23T19:05:07.344290Z\",\n", + " \"updateTime\": \"2021-03-23T19:05:07.344290Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "def get_latest_predictions(gcs_out_dir):\n", + " \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n", + " folders = !gsutil ls $gcs_out_dir\n", + " latest = \"\"\n", + " for folder in folders:\n", + " subfolder = folder.split(\"/\")[-2]\n", + " if subfolder.startswith(\"prediction-\"):\n", + " if subfolder > latest:\n", + " latest = folder[:-1]\n", + " return latest\n", + "\n", + "\n", + "while True:\n", + " response = clients[\"job\"].get_batch_prediction_job(name=batch_job_id)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", response.state)\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " folder = get_latest_predictions(\n", + " response.output_config.gcs_destination.output_uri_prefix\n", + " )\n", + " ! gsutil ls $folder/prediction*\n", + "\n", + " ! gsutil cat -h $folder/prediction*\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_get:migration,new,wait,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "==> gs://migration-ucaip-trainingaip-20210323185534/batch_output/prediction-custom_job_SKL20210323185534-2021_03_23T12_05_07_282Z/prediction.errors_stats-00000-of-00001 <==\n", + "\n", + "==> gs://migration-ucaip-trainingaip-20210323185534/batch_output/prediction-custom_job_SKL20210323185534-2021_03_23T12_05_07_282Z/prediction.results-00000-of-00001 <==\n", + "{\"instance\": [25, \"Private\", 226802, \"11th\", 7, \"Never-married\", \"Machine-op-inspct\", \"Own-child\", \"Black\", \"Male\", 0, 0, 40, \"United-States\"], \"prediction\": false}\n", + "{\"instance\": [38, \"Private\", 89814, \"HS-grad\", 9, \"Married-civ-spouse\", \"Farming-fishing\", \"Husband\", \"White\", \"Male\", 0, 0, 50, \"United-States\"], \"prediction\": false}\n", + "{\"instance\": [28, \"Local-gov\", 336951, \"Assoc-acdm\", 12, \"Married-civ-spouse\", \"Protective-serv\", \"Husband\", \"White\", \"Male\", 0, 0, 40, \"United-States\"], \"prediction\": false}\n", + "{\"instance\": [44, \"Private\", 160323, \"Some-college\", 10, \"Married-civ-spouse\", \"Machine-op-inspct\", \"Husband\", \"Black\", \"Male\", 7688, 0, 40, \"United-States\"], \"prediction\": true}\n", + "{\"instance\": [18, \"?\", 103497, \"Some-college\", 10, \"Never-married\", \"?\", \"Own-child\", \"White\", \"Female\", 0, 0, 30, \"United-States\"], \"prediction\": false}\n", + "{\"instance\": [34, \"Private\", 198693, \"10th\", 6, \"Never-married\", \"Other-service\", \"Not-in-family\", \"White\", \"Male\", 0, 0, 30, \"United-States\"], \"prediction\": false}\n", + "{\"instance\": [29, \"?\", 227026, \"HS-grad\", 9, \"Never-married\", \"?\", \"Unmarried\", \"Black\", \"Male\", 0, 0, 40, \"United-States\"], \"prediction\": false}\n", + "{\"instance\": [63, \"Self-emp-not-inc\", 104626, \"Prof-school\", 15, \"Married-civ-spouse\", \"Prof-specialty\", \"Husband\", \"White\", \"Male\", 3103, 0, 32, \"United-States\"], \"prediction\": false}\n", + "{\"instance\": [24, \"Private\", 369667, \"Some-college\", 10, \"Never-married\", \"Other-service\", \"Unmarried\", \"White\", \"Female\", 0, 0, 40, \"United-States\"], \"prediction\": false}\n", + "{\"instance\": [55, \"Private\", 104996, \"7th-8th\", 4, \"Married-civ-spouse\", \"Craft-repair\", \"Husband\", \"White\", \"Male\", 0, 0, 10, \"United-States\"], \"prediction\": false}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "be2ec9a417b1" + }, + "source": [ + "## Make online predictions" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_create:migration,new" + }, + "source": [ + "### [projects.locations.endpoints.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/create)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "request:migration" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_create:migration,new,request" + }, + "outputs": [], + "source": [ + "endpoint = {\"display_name\": \"custom_job_SKL\" + TIMESTAMP}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateEndpointRequest(parent=PARENT, endpoint=endpoint).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_create:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"endpoint\": {\n", + " \"displayName\": \"custom_job_SKL20210323185534\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "call:migration" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "response:migration" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,response" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_create:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/endpoints/695823734614786048\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_deploymodel:migration,new" + }, + "source": [ + "### [projects.locations.endpoints.deployModel](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/deployModel)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "request:migration" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_deploymodel:migration,new,request" + }, + "outputs": [], + "source": [ + "deployed_model = {\n", + " \"model\": model_id,\n", + " \"display_name\": \"custom_job_SKL\" + TIMESTAMP,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": 1,\n", + " \"max_replica_count\": 1,\n", + " \"machine_spec\": {\"machine_type\": \"n1-standard-4\", \"accelerator_count\": 0},\n", + " },\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.DeployModelRequest(\n", + " endpoint=endpoint_id,\n", + " deployed_model=deployed_model,\n", + " traffic_split={\"0\": 100},\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_deploymodel:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"endpoint\": \"projects/116273516712/locations/us-central1/endpoints/695823734614786048\",\n", + " \"deployedModel\": {\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/5984808915752189952\",\n", + " \"displayName\": \"custom_job_SKL20210323185534\",\n", + " \"dedicatedResources\": {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-4\"\n", + " },\n", + " \"minReplicaCount\": 1,\n", + " \"maxReplicaCount\": 1\n", + " }\n", + " },\n", + " \"trafficSplit\": {\n", + " \"0\": 100\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "call:migration" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_deploymodel:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint_id, deployed_model=deployed_model, traffic_split={\"0\": 100}\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "response:migration" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,response" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_deploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"deployedModel\": {\n", + " \"id\": \"6653241616695820288\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deployed_model_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The unique ID for the deployed model\n", + "deployed_model_id = result.deployed_model.id\n", + "\n", + "print(deployed_model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_predict:migration,new" + }, + "source": [ + "### [projects.locations.endpoints.predict](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/predict)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0d0bbb13eea3" + }, + "source": [ + "### Prepare file for online prediction" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "3b1abc212402" + }, + "outputs": [], + "source": [ + "INSTANCES = [\n", + " [\n", + " 25,\n", + " \"Private\",\n", + " 226802,\n", + " \"11th\",\n", + " 7,\n", + " \"Never-married\",\n", + " \"Machine-op-inspct\",\n", + " \"Own-child\",\n", + " \"Black\",\n", + " \"Male\",\n", + " 0,\n", + " 0,\n", + " 40,\n", + " \"United-States\",\n", + " ],\n", + " [\n", + " 38,\n", + " \"Private\",\n", + " 89814,\n", + " \"HS-grad\",\n", + " 9,\n", + " \"Married-civ-spouse\",\n", + " \"Farming-fishing\",\n", + " \"Husband\",\n", + " \"White\",\n", + " \"Male\",\n", + " 0,\n", + " 0,\n", + " 50,\n", + " \"United-States\",\n", + " ],\n", + " [\n", + " 28,\n", + " \"Local-gov\",\n", + " 336951,\n", + " \"Assoc-acdm\",\n", + " 12,\n", + " \"Married-civ-spouse\",\n", + " \"Protective-serv\",\n", + " \"Husband\",\n", + " \"White\",\n", + " \"Male\",\n", + " 0,\n", + " 0,\n", + " 40,\n", + " \"United-States\",\n", + " ],\n", + " [\n", + " 44,\n", + " \"Private\",\n", + " 160323,\n", + " \"Some-college\",\n", + " 10,\n", + " \"Married-civ-spouse\",\n", + " \"Machine-op-inspct\",\n", + " \"Husband\",\n", + " \"Black\",\n", + " \"Male\",\n", + " 7688,\n", + " 0,\n", + " 40,\n", + " \"United-States\",\n", + " ],\n", + " [\n", + " 18,\n", + " \"?\",\n", + " 103497,\n", + " \"Some-college\",\n", + " 10,\n", + " \"Never-married\",\n", + " \"?\",\n", + " \"Own-child\",\n", + " \"White\",\n", + " \"Female\",\n", + " 0,\n", + " 0,\n", + " 30,\n", + " \"United-States\",\n", + " ],\n", + " [\n", + " 34,\n", + " \"Private\",\n", + " 198693,\n", + " \"10th\",\n", + " 6,\n", + " \"Never-married\",\n", + " \"Other-service\",\n", + " \"Not-in-family\",\n", + " \"White\",\n", + " \"Male\",\n", + " 0,\n", + " 0,\n", + " 30,\n", + " \"United-States\",\n", + " ],\n", + " [\n", + " 29,\n", + " \"?\",\n", + " 227026,\n", + " \"HS-grad\",\n", + " 9,\n", + " \"Never-married\",\n", + " \"?\",\n", + " \"Unmarried\",\n", + " \"Black\",\n", + " \"Male\",\n", + " 0,\n", + " 0,\n", + " 40,\n", + " \"United-States\",\n", + " ],\n", + " [\n", + " 63,\n", + " \"Self-emp-not-inc\",\n", + " 104626,\n", + " \"Prof-school\",\n", + " 15,\n", + " \"Married-civ-spouse\",\n", + " \"Prof-specialty\",\n", + " \"Husband\",\n", + " \"White\",\n", + " \"Male\",\n", + " 3103,\n", + " 0,\n", + " 32,\n", + " \"United-States\",\n", + " ],\n", + " [\n", + " 24,\n", + " \"Private\",\n", + " 369667,\n", + " \"Some-college\",\n", + " 10,\n", + " \"Never-married\",\n", + " \"Other-service\",\n", + " \"Unmarried\",\n", + " \"White\",\n", + " \"Female\",\n", + " 0,\n", + " 0,\n", + " 40,\n", + " \"United-States\",\n", + " ],\n", + " [\n", + " 55,\n", + " \"Private\",\n", + " 104996,\n", + " \"7th-8th\",\n", + " 4,\n", + " \"Married-civ-spouse\",\n", + " \"Craft-repair\",\n", + " \"Husband\",\n", + " \"White\",\n", + " \"Male\",\n", + " 0,\n", + " 0,\n", + " 10,\n", + " \"United-States\",\n", + " ],\n", + "]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "request:migration" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_predict:migration,new,request,icn" + }, + "outputs": [], + "source": [ + "prediction_request = aip.PredictRequest(endpoint=endpoint_id)\n", + "prediction_request.instances.append(INSTANCES)\n", + "\n", + "print(MessageToJson(prediction_request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_deploymodel:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"endpoint\": \"projects/116273516712/locations/us-central1/endpoints/695823734614786048\",\n", + " \"instances\": [\n", + " [\n", + " [\n", + " 25.0,\n", + " \"Private\",\n", + " 226802.0,\n", + " \"11th\",\n", + " 7.0,\n", + " \"Never-married\",\n", + " \"Machine-op-inspct\",\n", + " \"Own-child\",\n", + " \"Black\",\n", + " \"Male\",\n", + " 0.0,\n", + " 0.0,\n", + " 40.0,\n", + " \"United-States\"\n", + " ],\n", + " [\n", + " 38.0,\n", + " \"Private\",\n", + " 89814.0,\n", + " \"HS-grad\",\n", + " 9.0,\n", + " \"Married-civ-spouse\",\n", + " \"Farming-fishing\",\n", + " \"Husband\",\n", + " \"White\",\n", + " \"Male\",\n", + " 0.0,\n", + " 0.0,\n", + " 50.0,\n", + " \"United-States\"\n", + " ],\n", + " [\n", + " 28.0,\n", + " \"Local-gov\",\n", + " 336951.0,\n", + " \"Assoc-acdm\",\n", + " 12.0,\n", + " \"Married-civ-spouse\",\n", + " \"Protective-serv\",\n", + " \"Husband\",\n", + " \"White\",\n", + " \"Male\",\n", + " 0.0,\n", + " 0.0,\n", + " 40.0,\n", + " \"United-States\"\n", + " ],\n", + " [\n", + " 44.0,\n", + " \"Private\",\n", + " 160323.0,\n", + " \"Some-college\",\n", + " 10.0,\n", + " \"Married-civ-spouse\",\n", + " \"Machine-op-inspct\",\n", + " \"Husband\",\n", + " \"Black\",\n", + " \"Male\",\n", + " 7688.0,\n", + " 0.0,\n", + " 40.0,\n", + " \"United-States\"\n", + " ],\n", + " [\n", + " 18.0,\n", + " \"?\",\n", + " 103497.0,\n", + " \"Some-college\",\n", + " 10.0,\n", + " \"Never-married\",\n", + " \"?\",\n", + " \"Own-child\",\n", + " \"White\",\n", + " \"Female\",\n", + " 0.0,\n", + " 0.0,\n", + " 30.0,\n", + " \"United-States\"\n", + " ],\n", + " [\n", + " 34.0,\n", + " \"Private\",\n", + " 198693.0,\n", + " \"10th\",\n", + " 6.0,\n", + " \"Never-married\",\n", + " \"Other-service\",\n", + " \"Not-in-family\",\n", + " \"White\",\n", + " \"Male\",\n", + " 0.0,\n", + " 0.0,\n", + " 30.0,\n", + " \"United-States\"\n", + " ],\n", + " [\n", + " 29.0,\n", + " \"?\",\n", + " 227026.0,\n", + " \"HS-grad\",\n", + " 9.0,\n", + " \"Never-married\",\n", + " \"?\",\n", + " \"Unmarried\",\n", + " \"Black\",\n", + " \"Male\",\n", + " 0.0,\n", + " 0.0,\n", + " 40.0,\n", + " \"United-States\"\n", + " ],\n", + " [\n", + " 63.0,\n", + " \"Self-emp-not-inc\",\n", + " 104626.0,\n", + " \"Prof-school\",\n", + " 15.0,\n", + " \"Married-civ-spouse\",\n", + " \"Prof-specialty\",\n", + " \"Husband\",\n", + " \"White\",\n", + " \"Male\",\n", + " 3103.0,\n", + " 0.0,\n", + " 32.0,\n", + " \"United-States\"\n", + " ],\n", + " [\n", + " 24.0,\n", + " \"Private\",\n", + " 369667.0,\n", + " \"Some-college\",\n", + " 10.0,\n", + " \"Never-married\",\n", + " \"Other-service\",\n", + " \"Unmarried\",\n", + " \"White\",\n", + " \"Female\",\n", + " 0.0,\n", + " 0.0,\n", + " 40.0,\n", + " \"United-States\"\n", + " ],\n", + " [\n", + " 55.0,\n", + " \"Private\",\n", + " 104996.0,\n", + " \"7th-8th\",\n", + " 4.0,\n", + " \"Married-civ-spouse\",\n", + " \"Craft-repair\",\n", + " \"Husband\",\n", + " \"White\",\n", + " \"Male\",\n", + " 0.0,\n", + " 0.0,\n", + " 10.0,\n", + " \"United-States\"\n", + " ]\n", + " ]\n", + " ]\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "call:migration" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_predict:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"prediction\"].predict(endpoint=endpoint_id, instances=INSTANCES)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "response:migration" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,request" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_predict:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"predictions\": [\n", + " false,\n", + " false,\n", + " false,\n", + " true,\n", + " false,\n", + " false,\n", + " false,\n", + " false,\n", + " false,\n", + " false\n", + " ],\n", + " \"deployedModelId\": \"6653241616695820288\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new" + }, + "source": [ + "### [projects.locations.endpoints.undeployModel](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/undeployModel)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "call:migration" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_undeploymodel:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint_id, deployed_model_id=deployed_model_id, traffic_split={}\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "response:migration" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,response" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup:migration,new" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup:migration,new" + }, + "outputs": [], + "source": [ + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_pipeline = True\n", + "delete_batchjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the model using the Vertex AI fully qualified identifier for the model\n", + "try:\n", + " if delete_model:\n", + " clients[\"model\"].delete_model(name=model_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex AI fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint:\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom training using the Vertex AI fully qualified identifier for the custome training\n", + "try:\n", + " if custom_training_id:\n", + " clients[\"job\"].delete_custom_job(name=custom_training_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex AI fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob:\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r gs://$BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "UJ10 unified Custom Training Prebuilt Container SKLearn.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/migration/UJ11 legacy HyperParameter Tuning Training Job with TensorFlow.ipynb b/notebooks/community/migration/UJ11 legacy HyperParameter Tuning Training Job with TensorFlow.ipynb new file mode 100644 index 000000000..f9d959b9b --- /dev/null +++ b/notebooks/community/migration/UJ11 legacy HyperParameter Tuning Training Job with TensorFlow.ipynb @@ -0,0 +1,1398 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title:migration,new" + }, + "source": [ + "# Vertex SDK: Submit a HyperParameter tuning training job with TensorFlow\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the Google *cloud-storage* library as well.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Y7Y8OgsQRlvq" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-storage" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the Kernel\n", + "\n", + "Once you've installed the Vertex SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "7LoqmDz3Rlvu" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "if not os.getenv(\"AUTORUN\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU run-time\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your GCP project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a GCP project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebooks.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend when possible, to choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You cannot use a Multi-Regional Storage bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see [Region support for Vertex services]()\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "XEwJmlbzRlvy" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "LwF83xbORlvz" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your GCP account\n", + "\n", + "**If you are using Google Cloud Notebooks**, your environment is already\n", + "authenticated. Skip this step.\n", + "\n", + "*Note: If you are on an Vertex notebook and run the cell, the cell knows to skip executing the authentication steps.*\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "MVoDIpoZRlv0" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your Google Cloud account. This provides access\n", + "# to your Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Vertex, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this tutorial in a notebook locally, replace the string\n", + " # below with the path to your service account key and run this cell to\n", + " # authenticate your Google Cloud account.\n", + " else:\n", + " %env GOOGLE_APPLICATION_CREDENTIALS your_path_to_credentials.json\n", + "\n", + " # Log in to your account on Google Cloud\n", + " ! gcloud auth login" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n", + " BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "vIi5JxUcRlv3" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "czQSbzsaRlv7" + }, + "outputs": [], + "source": [ + "! gsutil ls -al gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip" + }, + "source": [ + "#### Import Vertex SDK\n", + "\n", + "Import the Vertex SDK into our Python environment.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "-KblK5bSRlv8" + }, + "outputs": [], + "source": [ + "import json\n", + "import os\n", + "import sys\n", + "import time\n", + "from datetime import datetime\n", + "\n", + "from googleapiclient import discovery" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `PARENT`: The Vertex location root path for dataset, model and endpoint resources.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "terGI8TzRlv9" + }, + "outputs": [], + "source": [ + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients" + }, + "source": [ + "## Clients\n", + "\n", + "The Vertex SDK works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the server (Vertex).\n", + "\n", + "You will use several clients in this tutorial, so set them all up upfront.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "2o-kpasCRlv-" + }, + "outputs": [], + "source": [ + "client = discovery.build(\"ml\", \"v1\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "rI2IQP-rRlv_" + }, + "source": [ + "## Prepare a trainer script" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ldA8lSHtRlv_" + }, + "source": [ + "### Package assembly" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "5FT2-PifRlv_" + }, + "outputs": [], + "source": [ + "# Make folder for python training script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\\n", + "tag_build =\\n\\\n", + "tag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\\n", + "# Requires TensorFlow Datasets\\n\\\n", + "setuptools.setup(\\n\\\n", + " install_requires=[\\n\\\n", + " 'tensorflow_datasets==1.3.0',\\n\\\n", + " ],\\n\\\n", + " packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\\n", + "Name: Hyperparameter Tuning - Boston Housing\\n\\\n", + "Version: 0.0.0\\n\\\n", + "Summary: Demonstration hyperparameter tuning script\\n\\\n", + "Home-page: www.google.com\\n\\\n", + "Author: Google\\n\\\n", + "Author-email: aferlitsch@gmail.com\\n\\\n", + "License: Public\\n\\\n", + "Description: Demo\\n\\\n", + "Platform: Vertex AI\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ogDv9tBORlv_" + }, + "source": [ + "### Task.py contents" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "0-Bj8SCcRlwA" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Custom Training for Boston Housing\n", + " \n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "from hypertune import HyperTune\n", + "import numpy as np\n", + "import argparse\n", + "import os\n", + "import sys\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default='/tmp/saved_model', type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.001, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--units', dest='units',\n", + " default=64, type=int,\n", + " help='Number of units.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=20, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--param-file', dest='param_file',\n", + " default='/tmp/param.txt', type=str,\n", + " help='Output file for parameters')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "\n", + "def make_dataset():\n", + " # Scaling Boston Housing data features\n", + " def scale(feature):\n", + " max = np.max(feature)\n", + " feature = (feature / max).astype(np.float)\n", + " return feature, max\n", + "\n", + " (x_train, y_train), (x_test, y_test) = tf.keras.datasets.boston_housing.load_data(\n", + " path=\"boston_housing.npz\", test_split=0.2, seed=113\n", + " )\n", + " params = []\n", + " for _ in range(13):\n", + " x_train[_], max = scale(x_train[_])\n", + " x_test[_], _ = scale(x_test[_])\n", + " params.append(max)\n", + " \n", + " # store the normalization (max) value for each feature\n", + " with tf.io.gfile.GFile(args.param_file, 'w') as f:\n", + " f.write(str(params))\n", + " return (x_train, y_train), (x_test, y_test)\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_dnn_model():\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Dense(args.units, activation='relu', input_shape=(13,)),\n", + " tf.keras.layers.Dense(args.units, activation='relu'),\n", + " tf.keras.layers.Dense(1, activation='linear')\n", + " ])\n", + " model.compile(\n", + " loss='mse',\n", + " optimizer=tf.keras.optimizers.RMSprop(learning_rate=args.lr))\n", + " return model\n", + "\n", + "model = build_and_compile_dnn_model()\n", + "\n", + "# Instantiate the HyperTune reporting object\n", + "hpt = HyperTune()\n", + "\n", + "# Reporting callback\n", + "class HPTCallback(tf.keras.callbacks.Callback):\n", + "\n", + " def on_epoch_end(self, epoch, logs=None):\n", + " global hpt\n", + " hpt.report_hyperparameter_tuning_metric(\n", + " hyperparameter_metric_tag='val_loss',\n", + " metric_value=logs['val_loss'],\n", + " global_step=epoch)\n", + "\n", + "# Train the model\n", + "BATCH_SIZE = 16\n", + "(x_train, y_train), (x_test, y_test) = make_dataset()\n", + "model.fit(x_train, y_train, epochs=args.epochs, batch_size=BATCH_SIZE, validation_split=0.1, callbacks=[HPTCallback()])\n", + "model.save(args.model_dir)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "15pqExlERlwA" + }, + "source": [ + "### Store training script on your Cloud Storage bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "YsuaEFZuRlwB" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz gs://$BUCKET_NAME/hpt_boston_housing.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_create_and_deploy_model:migration" + }, + "source": [ + "## Train a model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0oqIBOSnJjkW" + }, + "source": [ + "### [projects.jobs.create](https://cloud.google.com/ai-platform/training/docs/reference/rest/v1/projects.jobs/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "TolZMkuSRlwC" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bmSJIBU3RlwC" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"hyperparameter_tuning_\" + TIMESTAMP\n", + "\n", + "training_input = {\n", + " \"scaleTier\": \"CUSTOM\",\n", + " \"masterType\": \"n1-standard-4\",\n", + " \"packageUris\": [\"gs://\" + BUCKET_NAME + \"/hpt_boston_housing.tar.gz\"],\n", + " \"pythonModule\": \"trainer.task\",\n", + " \"args\": [\n", + " \"--model-dir=\" + \"gs://{}/{}\".format(BUCKET_NAME, JOB_NAME),\n", + " ],\n", + " \"region\": REGION,\n", + " \"runtimeVersion\": \"2.1\",\n", + " \"pythonVersion\": \"3.7\",\n", + "}\n", + "\n", + "# Add hyperparameter tuning to the job config.\n", + "hyperparams = {\n", + " \"goal\": \"MINIMIZE\",\n", + " \"hyperparameterMetricTag\": \"val_loss\",\n", + " \"maxTrials\": 6,\n", + " \"maxParallelTrials\": 1,\n", + " \"params\": [],\n", + " \"algorithm\": \"RANDOM_SEARCH\",\n", + "}\n", + "\n", + "hyperparams[\"params\"].append(\n", + " {\n", + " \"parameterName\": \"lr\",\n", + " \"type\": \"DISCRETE\",\n", + " \"discreteValues\": [0.001, 0.01, 0.1],\n", + " \"scaleType\": \"UNIT_LINEAR_SCALE\",\n", + " }\n", + ")\n", + "\n", + "hyperparams[\"params\"].append(\n", + " {\n", + " \"parameterName\": \"units\",\n", + " \"type\": \"INTEGER\",\n", + " \"minValue\": 32,\n", + " \"maxValue\": 256,\n", + " \"scaleType\": \"UNIT_LINEAR_SCALE\",\n", + " }\n", + ")\n", + "\n", + "# Add hyperparameter specification to the training inputs dictionary.\n", + "training_input[\"hyperparameters\"] = hyperparams\n", + "\n", + "body = {\"jobId\": JOB_NAME, \"trainingInput\": training_input}\n", + "\n", + "request = (\n", + " client.projects()\n", + " .jobs()\n", + " .create(\n", + " parent=\"projects/\" + PROJECT_ID,\n", + " )\n", + ")\n", + "request.body = body\n", + "\n", + "print(json.dumps(json.loads(request.to_json()), indent=2))\n", + "\n", + "request = client.projects().jobs().create(parent=\"projects/\" + PROJECT_ID, body=body)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"uri\": \"https://ml.googleapis.com/v1/projects/migration-ucaip-training/jobs?alt=json\",\n", + " \"method\": \"POST\",\n", + " \"body\": {\n", + " \"jobId\": \"hyperparameter_tuning_20210226020553\",\n", + " \"trainingInput\": {\n", + " \"scaleTier\": \"CUSTOM\",\n", + " \"masterType\": \"n1-standard-4\",\n", + " \"packageUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210226020553/hpt_boston_housing.tar.gz\"\n", + " ],\n", + " \"pythonModule\": \"trainer.task\",\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210226020553/hyperparameter_tuning_20210226020553\"\n", + " ],\n", + " \"region\": \"us-central1\",\n", + " \"runtimeVersion\": \"2.1\",\n", + " \"pythonVersion\": \"3.7\",\n", + " \"hyperparameters\": {\n", + " \"goal\": \"MINIMIZE\",\n", + " \"hyperparameterMetricTag\": \"val_loss\",\n", + " \"maxTrials\": 6,\n", + " \"maxParallelTrials\": 1,\n", + " \"params\": [\n", + " {\n", + " \"parameterName\": \"lr\",\n", + " \"type\": \"DISCRETE\",\n", + " \"discreteValues\": [\n", + " 0.001,\n", + " 0.01,\n", + " 0.1\n", + " ],\n", + " \"scaleType\": \"UNIT_LINEAR_SCALE\"\n", + " },\n", + " {\n", + " \"parameterName\": \"units\",\n", + " \"type\": \"INTEGER\",\n", + " \"minValue\": 32,\n", + " \"maxValue\": 256,\n", + " \"scaleType\": \"UNIT_LINEAR_SCALE\"\n", + " }\n", + " ],\n", + " \"algorithm\": \"RANDOM_SEARCH\"\n", + " }\n", + " }\n", + " },\n", + " \"headers\": {\n", + " \"accept\": \"application/json\",\n", + " \"accept-encoding\": \"gzip, deflate\",\n", + " \"user-agent\": \"(gzip)\",\n", + " \"x-goog-api-client\": \"gdcl/1.12.8 gl-python/3.7.8\"\n", + " },\n", + " \"methodId\": \"ml.projects.jobs.create\",\n", + " \"resumable\": null,\n", + " \"response_callbacks\": [],\n", + " \"_in_error_state\": false,\n", + " \"body_size\": 0,\n", + " \"resumable_uri\": null,\n", + " \"resumable_progress\": 0\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wC08Kt4eRlwE" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "yNWyRHq_RlwE" + }, + "outputs": [], + "source": [ + "result = request.execute()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "rZZO8v7IRlwF" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "_dO8K_sDRlwF" + }, + "outputs": [], + "source": [ + "print(json.dumps(result, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "O85KoY2GRlwF" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"jobId\": \"hyperparameter_tuning_20210226020553\",\n", + " \"trainingInput\": {\n", + " \"scaleTier\": \"CUSTOM\",\n", + " \"masterType\": \"n1-standard-4\",\n", + " \"packageUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210226020553/hpt_boston_housing.tar.gz\"\n", + " ],\n", + " \"pythonModule\": \"trainer.task\",\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210226020553/hyperparameter_tuning_20210226020553\"\n", + " ],\n", + " \"hyperparameters\": {\n", + " \"goal\": \"MINIMIZE\",\n", + " \"params\": [\n", + " {\n", + " \"parameterName\": \"lr\",\n", + " \"type\": \"DISCRETE\",\n", + " \"discreteValues\": [\n", + " 0.001,\n", + " 0.01,\n", + " 0.1\n", + " ],\n", + " \"scaleType\": \"UNIT_LINEAR_SCALE\"\n", + " },\n", + " {\n", + " \"parameterName\": \"units\",\n", + " \"minValue\": 32,\n", + " \"maxValue\": 256,\n", + " \"type\": \"INTEGER\",\n", + " \"scaleType\": \"UNIT_LINEAR_SCALE\"\n", + " }\n", + " ],\n", + " \"maxTrials\": 6,\n", + " \"maxParallelTrials\": 1,\n", + " \"hyperparameterMetricTag\": \"val_loss\",\n", + " \"algorithm\": \"RANDOM_SEARCH\"\n", + " },\n", + " \"region\": \"us-central1\",\n", + " \"runtimeVersion\": \"2.1\",\n", + " \"pythonVersion\": \"3.7\"\n", + " },\n", + " \"createTime\": \"2021-02-26T02:06:01Z\",\n", + " \"state\": \"QUEUED\",\n", + " \"trainingOutput\": {\n", + " \"isHyperparameterTuningJob\": true\n", + " },\n", + " \"etag\": \"MPeWDTbQOUQ=\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "training_pipeline_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The short numeric ID for the custom training job\n", + "hyperparameter_tuning_short_id = result[\"jobId\"]\n", + "# The full unique ID for the custom training job\n", + "hyperparameter_tuning_id = \"projects/\" + PROJECT_ID + \"/jobs/\" + result[\"jobId\"]\n", + "\n", + "print(hyperparameter_tuning_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "PP31x5R4RlwG" + }, + "source": [ + "### [projects.jobs.get](https://cloud.google.com/ai-platform/training/docs/reference/rest/v1/projects.jobs/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "CNGyBG1iRlwG" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "mrlTlJw1RlwG" + }, + "outputs": [], + "source": [ + "request = client.projects().jobs().get(name=hyperparameter_tuning_id)\n", + "\n", + "result = request.execute()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dpGlbGq7RlwG" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "00k1rWSORlwG" + }, + "outputs": [], + "source": [ + "print(json.dumps(result, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "KzcY5CWlRlwH" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"jobId\": \"hyperparameter_tuning_20210226020553\",\n", + " \"trainingInput\": {\n", + " \"scaleTier\": \"CUSTOM\",\n", + " \"masterType\": \"n1-standard-4\",\n", + " \"packageUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210226020553/hpt_boston_housing.tar.gz\"\n", + " ],\n", + " \"pythonModule\": \"trainer.task\",\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210226020553/hyperparameter_tuning_20210226020553\"\n", + " ],\n", + " \"hyperparameters\": {\n", + " \"goal\": \"MINIMIZE\",\n", + " \"params\": [\n", + " {\n", + " \"parameterName\": \"lr\",\n", + " \"type\": \"DISCRETE\",\n", + " \"discreteValues\": [\n", + " 0.001,\n", + " 0.01,\n", + " 0.1\n", + " ],\n", + " \"scaleType\": \"UNIT_LINEAR_SCALE\"\n", + " },\n", + " {\n", + " \"parameterName\": \"units\",\n", + " \"minValue\": 32,\n", + " \"maxValue\": 256,\n", + " \"type\": \"INTEGER\",\n", + " \"scaleType\": \"UNIT_LINEAR_SCALE\"\n", + " }\n", + " ],\n", + " \"maxTrials\": 6,\n", + " \"maxParallelTrials\": 1,\n", + " \"hyperparameterMetricTag\": \"val_loss\",\n", + " \"algorithm\": \"RANDOM_SEARCH\"\n", + " },\n", + " \"region\": \"us-central1\",\n", + " \"runtimeVersion\": \"2.1\",\n", + " \"pythonVersion\": \"3.7\"\n", + " },\n", + " \"createTime\": \"2021-02-26T02:06:01Z\",\n", + " \"state\": \"QUEUED\",\n", + " \"trainingOutput\": {\n", + " \"isHyperparameterTuningJob\": true\n", + " },\n", + " \"etag\": \"GNEDjq+ds8I=\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "LWe8jWSTRlwH" + }, + "source": [ + "## Wait for the study to complete" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = client.projects().jobs().get(name=hyperparameter_tuning_id).execute()\n", + "\n", + " if response[\"state\"] != \"SUCCEEDED\":\n", + " print(\"Study trials have not completed:\", response[\"state\"])\n", + " if response[\"state\"] == \"FAILED\":\n", + " break\n", + " else:\n", + " print(\"Study trials have completed:\")\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cGHPc83JRlwH" + }, + "outputs": [], + "source": [ + "print(json.dumps(response, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "B5x1sVCCRlwI" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"jobId\": \"hyperparameter_tuning_20210226020553\",\n", + " \"trainingInput\": {\n", + " \"scaleTier\": \"CUSTOM\",\n", + " \"masterType\": \"n1-standard-4\",\n", + " \"packageUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210226020553/hpt_boston_housing.tar.gz\"\n", + " ],\n", + " \"pythonModule\": \"trainer.task\",\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210226020553/hyperparameter_tuning_20210226020553\"\n", + " ],\n", + " \"hyperparameters\": {\n", + " \"goal\": \"MINIMIZE\",\n", + " \"params\": [\n", + " {\n", + " \"parameterName\": \"lr\",\n", + " \"type\": \"DISCRETE\",\n", + " \"discreteValues\": [\n", + " 0.001,\n", + " 0.01,\n", + " 0.1\n", + " ],\n", + " \"scaleType\": \"UNIT_LINEAR_SCALE\"\n", + " },\n", + " {\n", + " \"parameterName\": \"units\",\n", + " \"minValue\": 32,\n", + " \"maxValue\": 256,\n", + " \"type\": \"INTEGER\",\n", + " \"scaleType\": \"UNIT_LINEAR_SCALE\"\n", + " }\n", + " ],\n", + " \"maxTrials\": 6,\n", + " \"maxParallelTrials\": 1,\n", + " \"hyperparameterMetricTag\": \"val_loss\",\n", + " \"algorithm\": \"RANDOM_SEARCH\"\n", + " },\n", + " \"region\": \"us-central1\",\n", + " \"runtimeVersion\": \"2.1\",\n", + " \"pythonVersion\": \"3.7\"\n", + " },\n", + " \"createTime\": \"2021-02-26T02:06:01Z\",\n", + " \"startTime\": \"2021-02-26T02:06:03Z\",\n", + " \"endTime\": \"2021-02-26T02:35:25Z\",\n", + " \"state\": \"SUCCEEDED\",\n", + " \"trainingOutput\": {\n", + " \"completedTrialCount\": \"6\",\n", + " \"trials\": [\n", + " {\n", + " \"trialId\": \"3\",\n", + " \"hyperparameters\": {\n", + " \"lr\": \"0.1\",\n", + " \"units\": \"39\"\n", + " },\n", + " \"finalMetric\": {\n", + " \"trainingStep\": \"8\",\n", + " \"objectiveValue\": 45.76672372585389\n", + " },\n", + " \"startTime\": \"2021-02-26T02:16:50.943369954Z\",\n", + " \"endTime\": \"2021-02-26T02:20:15Z\",\n", + " \"state\": \"SUCCEEDED\"\n", + " },\n", + " {\n", + " \"trialId\": \"5\",\n", + " \"hyperparameters\": {\n", + " \"lr\": \"0.1\",\n", + " \"units\": \"169\"\n", + " },\n", + " \"finalMetric\": {\n", + " \"trainingStep\": \"19\",\n", + " \"objectiveValue\": 46.09460951642292\n", + " },\n", + " \"startTime\": \"2021-02-26T02:26:23.876101420Z\",\n", + " \"endTime\": \"2021-02-26T02:29:59Z\",\n", + " \"state\": \"SUCCEEDED\"\n", + " },\n", + " {\n", + " \"trialId\": \"2\",\n", + " \"hyperparameters\": {\n", + " \"lr\": \"0.1\",\n", + " \"units\": \"69\"\n", + " },\n", + " \"finalMetric\": {\n", + " \"trainingStep\": \"19\",\n", + " \"objectiveValue\": 51.7642993461795\n", + " },\n", + " \"startTime\": \"2021-02-26T02:12:03.812631906Z\",\n", + " \"endTime\": \"2021-02-26T02:15:30Z\",\n", + " \"state\": \"SUCCEEDED\"\n", + " },\n", + " {\n", + " \"trialId\": \"1\",\n", + " \"hyperparameters\": {\n", + " \"lr\": \"0.01\",\n", + " \"units\": \"232\"\n", + " },\n", + " \"finalMetric\": {\n", + " \"trainingStep\": \"19\",\n", + " \"objectiveValue\": 55.58717541578339\n", + " },\n", + " \"startTime\": \"2021-02-26T02:06:40.790962827Z\",\n", + " \"endTime\": \"2021-02-26T02:10:47Z\",\n", + " \"state\": \"SUCCEEDED\"\n", + " },\n", + " {\n", + " \"trialId\": \"6\",\n", + " \"hyperparameters\": {\n", + " \"lr\": \"0.001\",\n", + " \"units\": \"123\"\n", + " },\n", + " \"finalMetric\": {\n", + " \"trainingStep\": \"19\",\n", + " \"objectiveValue\": 76.20946111911681\n", + " },\n", + " \"startTime\": \"2021-02-26T02:31:11.163541810Z\",\n", + " \"endTime\": \"2021-02-26T02:34:36Z\",\n", + " \"state\": \"SUCCEEDED\"\n", + " },\n", + " {\n", + " \"trialId\": \"4\",\n", + " \"hyperparameters\": {\n", + " \"lr\": \"0.01\",\n", + " \"units\": \"246\"\n", + " },\n", + " \"finalMetric\": {\n", + " \"trainingStep\": \"11\",\n", + " \"objectiveValue\": 100.75884600383479\n", + " },\n", + " \"startTime\": \"2021-02-26T02:21:36.635969861Z\",\n", + " \"endTime\": \"2021-02-26T02:25:02Z\",\n", + " \"state\": \"SUCCEEDED\"\n", + " }\n", + " ],\n", + " \"consumedMLUnits\": 0.39,\n", + " \"isHyperparameterTuningJob\": true,\n", + " \"hyperparameterMetricTag\": \"val_loss\"\n", + " },\n", + " \"etag\": \"eUgnylIe/+0=\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "1WqbO9gSRlwI" + }, + "source": [ + "## Review the results of the study" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "X0ooazLvRlwI" + }, + "outputs": [], + "source": [ + "best = (None, None, None, 0.0)\n", + "response = client.projects().jobs().get(name=hyperparameter_tuning_id).execute()\n", + "for trial in response[\"trainingOutput\"][\"trials\"]:\n", + " print(json.dumps(trial, indent=2))\n", + " # Keep track of the best outcome\n", + " try:\n", + " if float(trial.final_measurement.metrics[0].value) > best[3]:\n", + " best = (\n", + " trial.id,\n", + " float(trial.parameters[0].value),\n", + " float(trial.parameters[1].value),\n", + " float(trial.final_measurement.metrics[0].value),\n", + " )\n", + " except:\n", + " pass\n", + "\n", + "print()\n", + "print(\"ID\", best[0])\n", + "print(\"Decay\", best[1])\n", + "print(\"Learning Rate\", best[2])\n", + "print(\"Validation Accuracy\", best[3])" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "l4nWUlSIRlwI" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"trialId\": \"3\",\n", + " \"hyperparameters\": {\n", + " \"lr\": \"0.1\",\n", + " \"units\": \"39\"\n", + " },\n", + " \"finalMetric\": {\n", + " \"trainingStep\": \"8\",\n", + " \"objectiveValue\": 45.76672372585389\n", + " },\n", + " \"startTime\": \"2021-02-26T02:16:50.943369954Z\",\n", + " \"endTime\": \"2021-02-26T02:20:15Z\",\n", + " \"state\": \"SUCCEEDED\"\n", + "}\n", + "{\n", + " \"trialId\": \"5\",\n", + " \"hyperparameters\": {\n", + " \"lr\": \"0.1\",\n", + " \"units\": \"169\"\n", + " },\n", + " \"finalMetric\": {\n", + " \"trainingStep\": \"19\",\n", + " \"objectiveValue\": 46.09460951642292\n", + " },\n", + " \"startTime\": \"2021-02-26T02:26:23.876101420Z\",\n", + " \"endTime\": \"2021-02-26T02:29:59Z\",\n", + " \"state\": \"SUCCEEDED\"\n", + "}\n", + "{\n", + " \"trialId\": \"2\",\n", + " \"hyperparameters\": {\n", + " \"lr\": \"0.1\",\n", + " \"units\": \"69\"\n", + " },\n", + " \"finalMetric\": {\n", + " \"trainingStep\": \"19\",\n", + " \"objectiveValue\": 51.7642993461795\n", + " },\n", + " \"startTime\": \"2021-02-26T02:12:03.812631906Z\",\n", + " \"endTime\": \"2021-02-26T02:15:30Z\",\n", + " \"state\": \"SUCCEEDED\"\n", + "}\n", + "{\n", + " \"trialId\": \"1\",\n", + " \"hyperparameters\": {\n", + " \"lr\": \"0.01\",\n", + " \"units\": \"232\"\n", + " },\n", + " \"finalMetric\": {\n", + " \"trainingStep\": \"19\",\n", + " \"objectiveValue\": 55.58717541578339\n", + " },\n", + " \"startTime\": \"2021-02-26T02:06:40.790962827Z\",\n", + " \"endTime\": \"2021-02-26T02:10:47Z\",\n", + " \"state\": \"SUCCEEDED\"\n", + "}\n", + "{\n", + " \"trialId\": \"6\",\n", + " \"hyperparameters\": {\n", + " \"lr\": \"0.001\",\n", + " \"units\": \"123\"\n", + " },\n", + " \"finalMetric\": {\n", + " \"trainingStep\": \"19\",\n", + " \"objectiveValue\": 76.20946111911681\n", + " },\n", + " \"startTime\": \"2021-02-26T02:31:11.163541810Z\",\n", + " \"endTime\": \"2021-02-26T02:34:36Z\",\n", + " \"state\": \"SUCCEEDED\"\n", + "}\n", + "{\n", + " \"trialId\": \"4\",\n", + " \"hyperparameters\": {\n", + " \"lr\": \"0.01\",\n", + " \"units\": \"246\"\n", + " },\n", + " \"finalMetric\": {\n", + " \"trainingStep\": \"11\",\n", + " \"objectiveValue\": 100.75884600383479\n", + " },\n", + " \"startTime\": \"2021-02-26T02:21:36.635969861Z\",\n", + " \"endTime\": \"2021-02-26T02:25:02Z\",\n", + " \"state\": \"SUCCEEDED\"\n", + "}\n", + "\n", + "ID None\n", + "Decay None\n", + "Learning Rate None\n", + "Validation Accuracy 0.0\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup:migration,new" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "fyVK3PbnRlwJ" + }, + "outputs": [], + "source": [ + "delete_bucket = True\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r gs://$BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "collapsed_sections": [ + "import_aip", + "aip_constants", + "TolZMkuSRlwC", + "wC08Kt4eRlwE", + "rZZO8v7IRlwF", + "CNGyBG1iRlwG", + "dpGlbGq7RlwG" + ], + "name": "UJ11 legacy HyperParameter Tuning Training Job with TensorFlow.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/migration/UJ11 unified HyperParameter Tuning Training Job with TensorFlow.ipynb b/notebooks/community/migration/UJ11 unified HyperParameter Tuning Training Job with TensorFlow.ipynb new file mode 100644 index 000000000..54d7c4297 --- /dev/null +++ b/notebooks/community/migration/UJ11 unified HyperParameter Tuning Training Job with TensorFlow.ipynb @@ -0,0 +1,1373 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title:migration,new" + }, + "source": [ + "# Vertex SDK: Submit a HyperParameter tuning training job with TensorFlow\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest (preview) version of Vertex SDK.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "qXVD8TE-iBAZ" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-aiplatform --user" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the Google *cloud-storage* library as well.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "-JOoJeejiBAa" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-storage" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the Kernel\n", + "\n", + "Once you've installed the Vertex SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "SlJHybHWiBAa" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "if not os.getenv(\"AUTORUN\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU run-time\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your GCP project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a GCP project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebooks.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex AI. We recommend when possible, to choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You cannot use a Multi-Regional Storage bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see [Region support for Vertex AI services](https://cloud.google.com/vertex-ai/docs/general/locations)\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wkzC9Mn5iBAd" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "jX9n6pVLiBAd" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your GCP account\n", + "\n", + "**If you are using Google Cloud Notebooks**, your environment is already\n", + "authenticated. Skip this step.\n", + "\n", + "*Note: If you are on an Vertex notebook and run the cell, the cell knows to skip executing the authentication steps.*\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "BH7LjNZTiBAe" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your Google Cloud account. This provides access\n", + "# to your Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Vertex, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this tutorial in a notebook locally, replace the string\n", + " # below with the path to your service account key and run this cell to\n", + " # authenticate your Google Cloud account.\n", + " else:\n", + " %env GOOGLE_APPLICATION_CREDENTIALS your_path_to_credentials.json\n", + "\n", + " # Log in to your account on Google Cloud\n", + " ! gcloud auth login" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n", + " BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "IdMuD9HViBAf" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tRs6mJDwiBAg" + }, + "outputs": [], + "source": [ + "! gsutil ls -al gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip" + }, + "source": [ + "#### Import Vertex SDK\n", + "\n", + "Import the Vertex SDK into our Python environment.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "UCREN4OMiBAg" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex AI constants\n", + "\n", + "Setup up the following constants for Vertex AI:\n", + "\n", + "- `API_ENDPOINT`: The Vertex AI API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `API_PREDICT_ENDPOINT`: The Vertex AI API service endpoint for prediction.\n", + "- `PARENT`: The Vertex AI location root path for dataset, model and endpoint resources.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "A5-uS7XSiBAh" + }, + "outputs": [], + "source": [ + "# API Endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex AI location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients" + }, + "source": [ + "## Clients\n", + "\n", + "The Vertex SDK works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the server (Vertex).\n", + "\n", + "You will use several clients in this tutorial, so set them all up upfront.\n", + "\n", + "- Dataset Service for managed datasets.\n", + "- Model Service for managed models.\n", + "- Pipeline Service for training.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving. *Note*: Prediction has a different service endpoint.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "5DhFs5vNiBAi" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "clients[\"job\"] = create_job_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "7NnFMX6QiBAi" + }, + "source": [ + "## Prepare a trainer script" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "BrRN2X0DiBAi" + }, + "source": [ + "### Package assembly" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "rMP5OyIYiBAj" + }, + "outputs": [], + "source": [ + "# Make folder for python training script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\\n", + "tag_build =\\n\\\n", + "tag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\\n", + "# Requires TensorFlow Datasets\\n\\\n", + "setuptools.setup(\\n\\\n", + " install_requires=[\\n\\\n", + " 'tensorflow_datasets==1.3.0',\\n\\\n", + " ],\\n\\\n", + " packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\\n", + "Name: Hyperparameter Tuning - Boston Housing\\n\\\n", + "Version: 0.0.0\\n\\\n", + "Summary: Demonstration hyperparameter tuning script\\n\\\n", + "Home-page: www.google.com\\n\\\n", + "Author: Google\\n\\\n", + "Author-email: aferlitsch@gmail.com\\n\\\n", + "License: Public\\n\\\n", + "Description: Demo\\n\\\n", + "Platform: Vertex AI\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "nZLqlZ2OiBAj" + }, + "source": [ + "### Task.py contents" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "uDCJ3DXmiBAj" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# hyperparameter tuningfor Boston Housing\n", + " \n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "from hypertune import HyperTune\n", + "import numpy as np\n", + "import argparse\n", + "import os\n", + "import sys\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default='/tmp/saved_model', type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.001, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--units', dest='units',\n", + " default=64, type=int,\n", + " help='Number of units.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=20, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--param-file', dest='param_file',\n", + " default='/tmp/param.txt', type=str,\n", + " help='Output file for parameters')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "\n", + "def make_dataset():\n", + " # Scaling Boston Housing data features\n", + " def scale(feature):\n", + " max = np.max(feature)\n", + " feature = (feature / max).astype(np.float)\n", + " return feature, max\n", + "\n", + " (x_train, y_train), (x_test, y_test) = tf.keras.datasets.boston_housing.load_data(\n", + " path=\"boston_housing.npz\", test_split=0.2, seed=113\n", + " )\n", + " params = []\n", + " for _ in range(13):\n", + " x_train[_], max = scale(x_train[_])\n", + " x_test[_], _ = scale(x_test[_])\n", + " params.append(max)\n", + " \n", + " # store the normalization (max) value for each feature\n", + " with tf.io.gfile.GFile(args.param_file, 'w') as f:\n", + " f.write(str(params))\n", + " return (x_train, y_train), (x_test, y_test)\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_dnn_model():\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Dense(args.units, activation='relu', input_shape=(13,)),\n", + " tf.keras.layers.Dense(args.units, activation='relu'),\n", + " tf.keras.layers.Dense(1, activation='linear')\n", + " ])\n", + " model.compile(\n", + " loss='mse',\n", + " optimizer=tf.keras.optimizers.RMSprop(learning_rate=args.lr))\n", + " return model\n", + "\n", + "model = build_and_compile_dnn_model()\n", + "\n", + "# Instantiate the HyperTune reporting object\n", + "hpt = HyperTune()\n", + "\n", + "# Reporting callback\n", + "class HPTCallback(tf.keras.callbacks.Callback):\n", + "\n", + " def on_epoch_end(self, epoch, logs=None):\n", + " global hpt\n", + " hpt.report_hyperparameter_tuning_metric(\n", + " hyperparameter_metric_tag='val_loss',\n", + " metric_value=logs['val_loss'],\n", + " global_step=epoch)\n", + "\n", + "# Train the model\n", + "BATCH_SIZE = 16\n", + "(x_train, y_train), (x_test, y_test) = make_dataset()\n", + "model.fit(x_train, y_train, epochs=args.epochs, batch_size=BATCH_SIZE, validation_split=0.1, callbacks=[HPTCallback()])\n", + "model.save(args.model_dir)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "CDd81Xt8iBAj" + }, + "source": [ + "### Store training script on your Cloud Storage bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "8CofVaX6iBAk" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz gs://$BUCKET_NAME/hpt_boston_housing.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_create_and_deploy_model:migration" + }, + "source": [ + "## Train a model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0oqIBOSnJjkW" + }, + "source": [ + "### [projects.locations.hyperparameterTuningJob.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.hyperparameterTuningJobs/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "EILzZZ4biBAk" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "O2KyGzoYiBAk" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"hyperparameter_tuning_\" + TIMESTAMP\n", + "\n", + "WORKER_POOL_SPEC = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": {\"machine_type\": \"n1-standard-4\", \"accelerator_count\": 0},\n", + " \"python_package_spec\": {\n", + " \"executor_image_uri\": \"gcr.io/cloud-aiplatform/training/tf-cpu.2-1:latest\",\n", + " \"package_uris\": [\"gs://\" + BUCKET_NAME + \"/hpt_boston_housing.tar.gz\"],\n", + " \"python_module\": \"trainer.task\",\n", + " \"args\": [\"--model-dir=\" + \"gs://{}/{}\".format(BUCKET_NAME, JOB_NAME)],\n", + " },\n", + " }\n", + "]\n", + "\n", + "STUDY_SPEC = {\n", + " \"metrics\": [\n", + " {\"metric_id\": \"val_loss\", \"goal\": aip.StudySpec.MetricSpec.GoalType.MINIMIZE}\n", + " ],\n", + " \"parameters\": [\n", + " {\n", + " \"parameter_id\": \"lr\",\n", + " \"discrete_value_spec\": {\"values\": [0.001, 0.01, 0.1]},\n", + " \"scale_type\": aip.StudySpec.ParameterSpec.ScaleType.UNIT_LINEAR_SCALE,\n", + " },\n", + " {\n", + " \"parameter_id\": \"units\",\n", + " \"integer_value_spec\": {\"min_value\": 32, \"max_value\": 256},\n", + " \"scale_type\": aip.StudySpec.ParameterSpec.ScaleType.UNIT_LINEAR_SCALE,\n", + " },\n", + " ],\n", + " \"algorithm\": aip.StudySpec.Algorithm.RANDOM_SEARCH,\n", + "}\n", + "\n", + "hyperparameter_tuning_job = aip.HyperparameterTuningJob(\n", + " display_name=JOB_NAME,\n", + " trial_job_spec={\"worker_pool_specs\": WORKER_POOL_SPEC},\n", + " study_spec=STUDY_SPEC,\n", + " max_trial_count=6,\n", + " parallel_trial_count=1,\n", + ")\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateHyperparameterTuningJobRequest(\n", + " parent=PARENT, hyperparameter_tuning_job=hyperparameter_tuning_job\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"hyperparameterTuningJob\": {\n", + " \"displayName\": \"hyperparameter_tuning_20210226020029\",\n", + " \"studySpec\": {\n", + " \"metrics\": [\n", + " {\n", + " \"metricId\": \"val_loss\",\n", + " \"goal\": \"MINIMIZE\"\n", + " }\n", + " ],\n", + " \"parameters\": [\n", + " {\n", + " \"parameterId\": \"lr\",\n", + " \"discreteValueSpec\": {\n", + " \"values\": [\n", + " 0.001,\n", + " 0.01,\n", + " 0.1\n", + " ]\n", + " },\n", + " \"scaleType\": \"UNIT_LINEAR_SCALE\"\n", + " },\n", + " {\n", + " \"parameterId\": \"units\",\n", + " \"integerValueSpec\": {\n", + " \"minValue\": \"32\",\n", + " \"maxValue\": \"256\"\n", + " },\n", + " \"scaleType\": \"UNIT_LINEAR_SCALE\"\n", + " }\n", + " ],\n", + " \"algorithm\": \"RANDOM_SEARCH\"\n", + " },\n", + " \"maxTrialCount\": 6,\n", + " \"parallelTrialCount\": 1,\n", + " \"trialJobSpec\": {\n", + " \"workerPoolSpecs\": [\n", + " {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-4\"\n", + " },\n", + " \"replicaCount\": \"1\",\n", + " \"pythonPackageSpec\": {\n", + " \"executorImageUri\": \"gcr.io/cloud-aiplatform/training/tf-cpu.2-1:latest\",\n", + " \"packageUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210226020029/hpt_boston_housing.tar.gz\"\n", + " ],\n", + " \"pythonModule\": \"trainer.task\",\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210226020029/hyperparameter_tuning_20210226020029\"\n", + " ]\n", + " }\n", + " }\n", + " ]\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "94Kfzi6UiBAm" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "0NeORIoMiBAm" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].create_hyperparameter_tuning_job(\n", + " parent=PARENT, hyperparameter_tuning_job=hyperparameter_tuning_job\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "baTCgkvoiBAm" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cLTfBHC2iBAn" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "5ZY0QyWbiBAn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/hyperparameterTuningJobs/5264408897233354752\",\n", + " \"displayName\": \"hyperparameter_tuning_20210226020029\",\n", + " \"studySpec\": {\n", + " \"metrics\": [\n", + " {\n", + " \"metricId\": \"val_loss\",\n", + " \"goal\": \"MINIMIZE\"\n", + " }\n", + " ],\n", + " \"parameters\": [\n", + " {\n", + " \"parameterId\": \"lr\",\n", + " \"discreteValueSpec\": {\n", + " \"values\": [\n", + " 0.001,\n", + " 0.01,\n", + " 0.1\n", + " ]\n", + " },\n", + " \"scaleType\": \"UNIT_LINEAR_SCALE\"\n", + " },\n", + " {\n", + " \"parameterId\": \"units\",\n", + " \"integerValueSpec\": {\n", + " \"minValue\": \"32\",\n", + " \"maxValue\": \"256\"\n", + " },\n", + " \"scaleType\": \"UNIT_LINEAR_SCALE\"\n", + " }\n", + " ],\n", + " \"algorithm\": \"RANDOM_SEARCH\"\n", + " },\n", + " \"maxTrialCount\": 6,\n", + " \"parallelTrialCount\": 1,\n", + " \"trialJobSpec\": {\n", + " \"workerPoolSpecs\": [\n", + " {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-4\"\n", + " },\n", + " \"replicaCount\": \"1\",\n", + " \"diskSpec\": {\n", + " \"bootDiskType\": \"pd-ssd\",\n", + " \"bootDiskSizeGb\": 100\n", + " },\n", + " \"pythonPackageSpec\": {\n", + " \"executorImageUri\": \"gcr.io/cloud-aiplatform/training/tf-cpu.2-1:latest\",\n", + " \"packageUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210226020029/hpt_boston_housing.tar.gz\"\n", + " ],\n", + " \"pythonModule\": \"trainer.task\",\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210226020029/hyperparameter_tuning_20210226020029\"\n", + " ]\n", + " }\n", + " }\n", + " ]\n", + " },\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"createTime\": \"2021-02-26T02:02:02.787187Z\",\n", + " \"updateTime\": \"2021-02-26T02:02:02.787187Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "training_pipeline_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the hyperparameter tuningjob\n", + "hyperparameter_tuning_id = request.name\n", + "# The short numeric ID for the hyperparameter tuningjob\n", + "hyperparameter_tuning_short_id = hyperparameter_tuning_id.split(\"/\")[-1]\n", + "\n", + "print(hyperparameter_tuning_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "uNbWmXHziBAo" + }, + "source": [ + "### [projects.locations.hyperparameterTuningJob.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.hyperparameterTuningJobs/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "TOP1v7ybiBAo" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "CKH3m0NTiBAo" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].get_hyperparameter_tuning_job(name=hyperparameter_tuning_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "FVgPQi7MiBAo" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "P0EdlOAeiBAp" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "zJ7nxJ2OiBAp" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/hyperparameterTuningJobs/5264408897233354752\",\n", + " \"displayName\": \"hyperparameter_tuning_20210226020029\",\n", + " \"studySpec\": {\n", + " \"metrics\": [\n", + " {\n", + " \"metricId\": \"val_loss\",\n", + " \"goal\": \"MINIMIZE\"\n", + " }\n", + " ],\n", + " \"parameters\": [\n", + " {\n", + " \"parameterId\": \"lr\",\n", + " \"discreteValueSpec\": {\n", + " \"values\": [\n", + " 0.001,\n", + " 0.01,\n", + " 0.1\n", + " ]\n", + " },\n", + " \"scaleType\": \"UNIT_LINEAR_SCALE\"\n", + " },\n", + " {\n", + " \"parameterId\": \"units\",\n", + " \"integerValueSpec\": {\n", + " \"minValue\": \"32\",\n", + " \"maxValue\": \"256\"\n", + " },\n", + " \"scaleType\": \"UNIT_LINEAR_SCALE\"\n", + " }\n", + " ],\n", + " \"algorithm\": \"RANDOM_SEARCH\"\n", + " },\n", + " \"maxTrialCount\": 6,\n", + " \"parallelTrialCount\": 1,\n", + " \"trialJobSpec\": {\n", + " \"workerPoolSpecs\": [\n", + " {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-4\"\n", + " },\n", + " \"replicaCount\": \"1\",\n", + " \"diskSpec\": {\n", + " \"bootDiskType\": \"pd-ssd\",\n", + " \"bootDiskSizeGb\": 100\n", + " },\n", + " \"pythonPackageSpec\": {\n", + " \"executorImageUri\": \"gcr.io/cloud-aiplatform/training/tf-cpu.2-1:latest\",\n", + " \"packageUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210226020029/hpt_boston_housing.tar.gz\"\n", + " ],\n", + " \"pythonModule\": \"trainer.task\",\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210226020029/hyperparameter_tuning_20210226020029\"\n", + " ]\n", + " }\n", + " }\n", + " ]\n", + " },\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"createTime\": \"2021-02-26T02:02:02.787187Z\",\n", + " \"updateTime\": \"2021-02-26T02:02:02.787187Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "QflEzRwkiBAp" + }, + "source": [ + "## Wait for the study to complete" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = clients[\"job\"].get_hyperparameter_tuning_job(\n", + " name=hyperparameter_tuning_id\n", + " )\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Study trials have not completed:\", response.state)\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " break\n", + " else:\n", + " print(\"Study trials have completed:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(20)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cL8hxrQ6iBAp" + }, + "source": [ + "## Review the results of the study" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "blbKukdkiBAq" + }, + "outputs": [], + "source": [ + "best = (None, None, None, 0.0)\n", + "response = clients[\"job\"].get_hyperparameter_tuning_job(name=hyperparameter_tuning_id)\n", + "for trial in response.trials:\n", + " print(MessageToJson(trial.__dict__[\"_pb\"]))\n", + " # Keep track of the best outcome\n", + " try:\n", + " if float(trial.final_measurement.metrics[0].value) > best[3]:\n", + " best = (\n", + " trial.id,\n", + " float(trial.parameters[0].value),\n", + " float(trial.parameters[1].value),\n", + " float(trial.final_measurement.metrics[0].value),\n", + " )\n", + " except:\n", + " pass\n", + "\n", + "print()\n", + "print(\"ID\", best[0])\n", + "print(\"Decay\", best[1])\n", + "print(\"Learning Rate\", best[2])\n", + "print(\"Validation Accuracy\", best[3])" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "4ONNEM15iBAq" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"id\": \"1\",\n", + " \"state\": \"SUCCEEDED\",\n", + " \"parameters\": [\n", + " {\n", + " \"parameterId\": \"lr\",\n", + " \"value\": 0.1\n", + " },\n", + " {\n", + " \"parameterId\": \"units\",\n", + " \"value\": 80.0\n", + " }\n", + " ],\n", + " \"finalMeasurement\": {\n", + " \"stepCount\": \"19\",\n", + " \"metrics\": [\n", + " {\n", + " \"metricId\": \"val_loss\",\n", + " \"value\": 46.61515110294993\n", + " }\n", + " ]\n", + " },\n", + " \"startTime\": \"2021-02-26T02:05:16.935353384Z\",\n", + " \"endTime\": \"2021-02-26T02:12:44Z\"\n", + "}\n", + "{\n", + " \"id\": \"2\",\n", + " \"state\": \"SUCCEEDED\",\n", + " \"parameters\": [\n", + " {\n", + " \"parameterId\": \"lr\",\n", + " \"value\": 0.01\n", + " },\n", + " {\n", + " \"parameterId\": \"units\",\n", + " \"value\": 45.0\n", + " }\n", + " ],\n", + " \"finalMeasurement\": {\n", + " \"stepCount\": \"19\",\n", + " \"metrics\": [\n", + " {\n", + " \"metricId\": \"val_loss\",\n", + " \"value\": 32.55313952376203\n", + " }\n", + " ]\n", + " },\n", + " \"startTime\": \"2021-02-26T02:15:31.357856840Z\",\n", + " \"endTime\": \"2021-02-26T02:24:18Z\"\n", + "}\n", + "{\n", + " \"id\": \"3\",\n", + " \"state\": \"SUCCEEDED\",\n", + " \"parameters\": [\n", + " {\n", + " \"parameterId\": \"lr\",\n", + " \"value\": 0.1\n", + " },\n", + " {\n", + " \"parameterId\": \"units\",\n", + " \"value\": 70.0\n", + " }\n", + " ],\n", + " \"finalMeasurement\": {\n", + " \"stepCount\": \"19\",\n", + " \"metrics\": [\n", + " {\n", + " \"metricId\": \"val_loss\",\n", + " \"value\": 42.709188321741614\n", + " }\n", + " ]\n", + " },\n", + " \"startTime\": \"2021-02-26T02:26:40.704476222Z\",\n", + " \"endTime\": \"2021-02-26T02:34:21Z\"\n", + "}\n", + "{\n", + " \"id\": \"4\",\n", + " \"state\": \"SUCCEEDED\",\n", + " \"parameters\": [\n", + " {\n", + " \"parameterId\": \"lr\",\n", + " \"value\": 0.01\n", + " },\n", + " {\n", + " \"parameterId\": \"units\",\n", + " \"value\": 173.0\n", + " }\n", + " ],\n", + " \"finalMeasurement\": {\n", + " \"stepCount\": \"17\",\n", + " \"metrics\": [\n", + " {\n", + " \"metricId\": \"val_loss\",\n", + " \"value\": 46.12480219399057\n", + " }\n", + " ]\n", + " },\n", + " \"startTime\": \"2021-02-26T02:37:45.275581053Z\",\n", + " \"endTime\": \"2021-02-26T02:51:07Z\"\n", + "}\n", + "{\n", + " \"id\": \"5\",\n", + " \"state\": \"SUCCEEDED\",\n", + " \"parameters\": [\n", + " {\n", + " \"parameterId\": \"lr\",\n", + " \"value\": 0.01\n", + " },\n", + " {\n", + " \"parameterId\": \"units\",\n", + " \"value\": 223.0\n", + " }\n", + " ],\n", + " \"finalMeasurement\": {\n", + " \"stepCount\": \"19\",\n", + " \"metrics\": [\n", + " {\n", + " \"metricId\": \"val_loss\",\n", + " \"value\": 24.875632611716664\n", + " }\n", + " ]\n", + " },\n", + " \"startTime\": \"2021-02-26T02:53:32.612612421Z\",\n", + " \"endTime\": \"2021-02-26T02:54:19Z\"\n", + "}\n", + "{\n", + " \"id\": \"6\",\n", + " \"state\": \"SUCCEEDED\",\n", + " \"parameters\": [\n", + " {\n", + " \"parameterId\": \"lr\",\n", + " \"value\": 0.1\n", + " },\n", + " {\n", + " \"parameterId\": \"units\",\n", + " \"value\": 123.0\n", + " }\n", + " ],\n", + " \"finalMeasurement\": {\n", + " \"stepCount\": \"13\",\n", + " \"metrics\": [\n", + " {\n", + " \"metricId\": \"val_loss\",\n", + " \"value\": 43.352300690441595\n", + " }\n", + " ]\n", + " },\n", + " \"startTime\": \"2021-02-26T02:56:47.323707459Z\",\n", + " \"endTime\": \"2021-02-26T03:03:49Z\"\n", + "}\n", + "\n", + "ID 1\n", + "Decay 0.1\n", + "Learning Rate 80.0\n", + "Validation Accuracy 46.61515110294993\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup:migration,new" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "QACczMumiBAq" + }, + "outputs": [], + "source": [ + "delete_hpt_job = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the hyperparameter tuningusing the Vertex AI fully qualified identifier for the custome training\n", + "try:\n", + " if delete_hpt_job:\n", + " clients[\"job\"].delete_hyperparameter_tuning_job(name=hyperparameter_tuning_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r gs://$BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "UJ11 unified HyperParameter Tuning Training Job with TensorFlow.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/migration/UJ13 legacy Data Labeling task.ipynb b/notebooks/community/migration/UJ13 legacy Data Labeling task.ipynb new file mode 100644 index 000000000..ead46f460 --- /dev/null +++ b/notebooks/community/migration/UJ13 legacy Data Labeling task.ipynb @@ -0,0 +1,1485 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License.\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title:migration,new" + }, + "source": [ + "# AutoML SDK: Data Labeling\n", + "\n", + "*Disclaimer*: This notebook is for illustrative purposes only." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest (preview) version of AutoML SDK.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "eGO6_2SExXiI" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-automl --user\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-datalabeling --user\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the Google *cloud-storage* library as well.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "G0bLTzVWxXiJ" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-storage\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the Kernel\n", + "\n", + "Once you've installed the AutoML SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "33Uq-fKkxXiK" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "\n", + "if not os.getenv(\"AUTORUN\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU run-time\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your GCP project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a GCP project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the AutoML APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [Google Cloud SDK](https://cloud.google.com/sdk) is already installed in AutoML Notebooks.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" #@param {type:\"string\"}\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id", + "outputId": "2663b1ca-f6c6-43ed-c23b-654919cb200c" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for AutoML. We recommend when possible, to choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You cannot use a Multi-Regional Storage bucket for training with AutoML. Not all regions provide support for all AutoML services. For the latest support per region, see [Region support for AutoML services]()\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "_eH8WcWTxXiM" + }, + "outputs": [], + "source": [ + "REGION = 'us-central1' #@param {type: \"string\"}\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "NoqPBceExXiM" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your GCP account\n", + "\n", + "**If you are using AutoML Notebooks**, your environment is already\n", + "authenticated. Skip this step.\n", + "\n", + "*Note: If you are on an AutoML notebook and run the cell, the cell knows to skip executing the authentication steps.*\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "6aUuFIftxXiN" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your Google Cloud account. This provides access\n", + "# to your Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on AutoML, then don't execute this code\n", + "if not os.path.exists('/opt/deeplearning/metadata/env_version'):\n", + " if 'google.colab' in sys.modules:\n", + " from google.colab import auth as google_auth\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this tutorial in a notebook locally, replace the string\n", + " # below with the path to your service account key and run this cell to\n", + " # authenticate your Google Cloud account.\n", + " else:\n", + " %env GOOGLE_APPLICATION_CREDENTIALS your_path_to_credentials.json\n", + "\n", + " # Log in to your account on Google Cloud\n", + " ! gcloud auth login\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"[your-bucket-name]\" #@param {type:\"string\"}\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n", + " BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "-8qURIGmxXiO", + "outputId": "e887810d-a0fd-4063-d636-20313733b955" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION gs://$BUCKET_NAME\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "t_65JAARxXiP" + }, + "outputs": [], + "source": [ + "! gsutil ls -al gs://$BUCKET_NAME\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip" + }, + "source": [ + "#### Import AutoML SDK\n", + "\n", + "Import the AutoML SDK into our Python environment.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "P8L1tKAexXiQ" + }, + "outputs": [], + "source": [ + "import json\n", + "import os\n", + "import sys\n", + "import time\n", + "\n", + "\n", + "from google.cloud import automl\n", + "from google.cloud import datalabeling_v1beta1 as datalabeling\n", + "\n", + "\n", + "from google.protobuf.json_format import MessageToJson\n", + "from google.protobuf.json_format import ParseDict\n", + "from googleapiclient.discovery import build\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Setup up the following constants for AutoML:\n", + "\n", + "- `PARENT`: The AutoML location root path for dataset, model and endpoint resources.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "rG1L9BEzxXiR" + }, + "outputs": [], + "source": [ + "# AutoML location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients" + }, + "source": [ + "## Clients\n", + "\n", + "The AutoML SDK works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the server (AutoML).\n", + "\n", + "You will use several clients in this tutorial, so set them all up upfront.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "c3sR3OxmxXiR", + "outputId": "5bc213e7-2df8-4aca-dbf8-f8d267e98d52" + }, + "outputs": [], + "source": [ + "def data_labeling_client():\n", + " return datalabeling.DataLabelingServiceClient()\n", + "\n", + "\n", + "def operations_client():\n", + " return datalabeling.DataLabelingServiceClient()._transport.operations_client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"labeling\"] = data_labeling_client()\n", + "clients[\"operations\"] = operations_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:flowers,csv,icn" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "\n", + "LABELING_FILES = [\n", + " \"https://raw.githubusercontent.com/googleapis/python-aiplatform/master/samples/snippets/resources/daisy.jpg\"\n", + "]\n", + "\n", + "IMPORT_FILE = \"gs://\" + BUCKET_NAME + '/labeling.csv'\n", + "with tf.io.gfile.GFile(IMPORT_FILE, 'w') as f:\n", + " for lf in LABELING_FILES:\n", + " ! wget {lf} | gsutil cp {lf.split(\"/\")[-1]} gs://{BUCKET_NAME}\n", + " f.write(\"gs://\" + BUCKET_NAME + \"/\" + lf.split(\"/\")[-1] + \"\\n\")\n", + " " + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cBgFrDD1_i__", + "outputId": "73392b9a-b0b3-4713-f768-c1e0ae4c48fb" + }, + "outputs": [], + "source": [ + "! gsutil cat $IMPORT_FILE\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_create:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210303171756/daisy.jpg\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_a_dataset:migration" + }, + "source": [ + "## Create a dataset\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bkdu67d_xXiT" + }, + "source": [ + "### [projects.locations.datasets.create](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.datasets/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "request:migration" + }, + "source": [ + "#### Request\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "uTrBPMR5xXiU", + "outputId": "8da73a34-a13f-4e1f-c8af-7f25338db142" + }, + "outputs": [], + "source": [ + "dataset = {\n", + " \"display_name\": \"labeling_\" + TIMESTAMP,\n", + " \"description\": \"labeling_\" + TIMESTAMP\n", + "}\n", + "\n", + "print(MessageToJson(\n", + " datalabeling.CreateDatasetRequest(\n", + " parent=PARENT,\n", + " dataset=dataset,\n", + " ).__dict__[\"_pb\"])\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "jvhVeS_ExXiU" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"dataset\": {\n", + " \"displayName\": \"labeling_20210303171756\",\n", + " \"description\": \"labeling_20210303171756\"\n", + " }\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "call:migration" + }, + "source": [ + "#### Call\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "datasets_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"labeling\"].create_dataset(\n", + " parent=PARENT,\n", + " dataset=dataset\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "response:migration" + }, + "source": [ + "#### Response\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,response", + "outputId": "addad607-2542-45c2-d1b1-5ca7aa1d5d0c" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_create:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/migration-ucaip-training/datasets/60401f09_0000_2cac_bcb5_3c286d3b27b6\",\n", + " \"displayName\": \"labeling_20210303171756\",\n", + " \"description\": \"labeling_20210303171756\",\n", + " \"createTime\": \"2021-03-04T13:13:20.227435060Z\"\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:migration,new,response", + "outputId": "d201ed62-717f-4798-9006-a299684d4bf5" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = request.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split('/')[-1]\n", + "\n", + "print(dataset_id)\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "qHWi_CMnxXiV" + }, + "source": [ + "### [projects.locations.datasets.importData](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.datasets/importData)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "py-0niVqxXiW" + }, + "source": [ + "#### Request\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "datasets_import:migration,new,request", + "outputId": "83bf0203-7782-42e8-ab78-a23eb1679561" + }, + "outputs": [], + "source": [ + "input_config = {\n", + " \"data_type\": \"IMAGE\",\n", + " \"gcs_source\": {\n", + " \"input_uri\": IMPORT_FILE, \n", + " \"mime_type\": \"text/csv\"\n", + " }\n", + "}\n", + "\n", + "print(MessageToJson(\n", + " datalabeling.ImportDataRequest(\n", + " name=dataset_id,\n", + " input_config=input_config\n", + " ).__dict__[\"_pb\"])\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cxbp1cGzxXiW" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/migration-ucaip-training/datasets/60401f09_0000_2cac_bcb5_3c286d3b27b6\",\n", + " \"inputConfig\": {\n", + " \"dataType\": \"IMAGE\",\n", + " \"gcsSource\": {\n", + " \"inputUri\": \"gs://migration-ucaip-trainingaip-20210303171756/labeling.csv\",\n", + " \"mimeType\": \"text/csv\"\n", + " }\n", + " }\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "v8kGJFwYxXiW" + }, + "source": [ + "#### Call\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "datasets_import:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"labeling\"].import_data(\n", + " name=dataset_id,\n", + " input_config=input_config\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "UxcyIvGLxXiW" + }, + "source": [ + "#### Response\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "zXlYWLR7xXiX", + "outputId": "d5740c78-a8db-4d69-84d7-72cf3b83cf87" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_a_model:migration" + }, + "source": [ + "## Create data labeling job\n", + "Following methods in this section are for illustrative purposes only.\n" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### [projects.annotationSpecSets.create](https://cloud.google.com/ai-platform/data-labeling/docs/reference/rest/v1beta1/projects.annotationSpecSets/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bytx9mM0xXiX" + }, + "source": [ + "#### Request\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "annotation_spec_set = {\n", + " \"display_name\": \"labeling_\" + TIMESTAMP,\n", + " \"description\": \"description\",\n", + " \"annotation_specs\": [\n", + " {\n", + " \"display_name\": \"rose\",\n", + " \"description\": \"rose description\"\n", + " \n", + " }\n", + " ]\n", + "}\n", + "\n", + "print(MessageToJson(\n", + " datalabeling.CreateAnnotationSpecSetRequest(\n", + " parent=\"projects/\" + PROJECT_ID,\n", + " annotation_spec_set=annotation_spec_set\n", + " ).__dict__[\"_pb\"])\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cxbp1cGzxXiW" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training\",\n", + " \"annotationSpecSet\": {\n", + " \"displayName\": \"labeling_20210303171756\",\n", + " \"description\": \"description\",\n", + " \"annotationSpecs\": [\n", + " {\n", + " \"displayName\": \"rose\",\n", + " \"description\": \"rose description\"\n", + " }\n", + " ]\n", + " }\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cNNI6bjcxXiY" + }, + "source": [ + "#### Call\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "request = clients[\"labeling\"].create_annotation_spec_set(\n", + " parent=\"projects/\" + PROJECT_ID,\n", + " annotation_spec_set=annotation_spec_set\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "UxcyIvGLxXiW" + }, + "source": [ + "#### Response\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cxbp1cGzxXiW" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/migration-ucaip-training/annotationSpecSets/60401d46_0000_2c8d_aa13_883d24f7e3c8\",\n", + " \"displayName\": \"labeling_20210303171756\",\n", + " \"description\": \"description\",\n", + " \"annotationSpecs\": [\n", + " {\n", + " \"displayName\": \"rose\",\n", + " \"description\": \"rose description\"\n", + " }\n", + " ]\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "annotation_spec_set_name = request.name\n", + "\n", + "print(annotation_spec_set_name)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "### [projects.instructions.create](https://cloud.google.com/ai-platform/data-labeling/docs/reference/rest/v1beta1/projects.instructions/create)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# create placeholder file for valid PDF file with instruction for data labeling\n", + "! echo \"this is instruction\" >> instruction.txt | gsutil cp instruction.txt gs://$BUCKET_NAME \n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bytx9mM0xXiX" + }, + "source": [ + "#### Request\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "INSTRUCTION_FILE = \"gs://\" + BUCKET_NAME + \"/instruction.txt\" \n", + "\n", + "instruction = {\n", + " \"display_name\": \"labeling_\" + TIMESTAMP,\n", + " \"description\": \"description\",\n", + " \"data_type\": \"IMAGE\",\n", + " \"pdf_instruction\": {\n", + " \"gcs_file_uri\": INSTRUCTION_FILE\n", + " }\n", + "}\n", + "\n", + "print(MessageToJson(\n", + " datalabeling.CreateInstructionRequest(\n", + " parent=\"projects/\" + PROJECT_ID,\n", + " instruction=instruction\n", + " ).__dict__[\"_pb\"])\n", + ")\n", + " " + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cxbp1cGzxXiW" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training\",\n", + " \"instruction\": {\n", + " \"displayName\": \"labeling_20210303171756\",\n", + " \"description\": \"description\",\n", + " \"dataType\": \"IMAGE\",\n", + " \"pdfInstruction\": {\n", + " \"gcsFileUri\": \"gs://migration-ucaip-trainingaip-20210303171756/instruction.txt\"\n", + " }\n", + " }\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cNNI6bjcxXiY" + }, + "source": [ + "#### Call\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "request = clients[\"labeling\"].create_instruction(\n", + " parent=\"projects/\" + PROJECT_ID,\n", + " instruction=instruction\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "UxcyIvGLxXiW" + }, + "source": [ + "#### Response\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cxbp1cGzxXiW" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/migration-ucaip-training/instructions/60401ef8_0000_2cac_bcb5_3c286d3b27b6\",\n", + " \"displayName\": \"labeling_20210303171756\",\n", + " \"description\": \"description\",\n", + " \"createTime\": \"1970-01-01T00:00:00Z\",\n", + " \"dataType\": \"IMAGE\",\n", + " \"pdfInstruction\": {\n", + " \"gcsFileUri\": \"gs://migration-ucaip-trainingaip-20210303171756/instruction.txt\"\n", + " }\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "instruction_name = result.name\n", + "\n", + "print(instruction_name)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "yfufMwAEJjkX" + }, + "source": [ + "### [projects.datasets.image.label](https://cloud.google.com/ai-platform/data-labeling/docs/reference/rest/v1beta1/projects.datasets.image/label)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bytx9mM0xXiX" + }, + "source": [ + "#### Request\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "EMAIL = \"dev@fourteen33.com\"\n", + "\n", + "basic_config = {\n", + " \"instruction\": instruction_name,\n", + " \"annotated_dataset_display_name\": \"labeling_\" + TIMESTAMP,\n", + " \"label_group\": \"rose\",\n", + " \"replica_count\":1,\n", + " \"contributor_emails\": [EMAIL]\n", + "}\n", + "\n", + "feature = \"CLASSIFICATION\"\n", + "\n", + "config = {\n", + " \"annotation_spec_set\": annotation_spec_set_name,\n", + " \"allow_multi_label\": False,\n", + " \"answer_aggregation_type\": \"MAJORITY_VOTE\"\n", + "}\n", + "\n", + "print(MessageToJson(\n", + " datalabeling.LabelImageRequest(\n", + " parent=dataset_id,\n", + " basic_config=basic_config,\n", + " feature=feature,\n", + " image_classification_config=config \n", + " ).__dict__[\"_pb\"])\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cxbp1cGzxXiW" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1/datasets/60401f09_0000_2cac_bcb5_3c286d3b27b6\",\n", + " \"basicConfig\": {\n", + " \"instruction\": \"projects/migration-ucaip-training/instructions/60401ef8_0000_2cac_bcb5_3c286d3b27b6\",\n", + " \"annotatedDatasetDisplayName\": \"labeling_20210303171756\",\n", + " \"labelGroup\": \"rose\",\n", + " \"replicaCount\": 1,\n", + " \"contributorEmails\": [\n", + " \"dev@fourteen33.com\"\n", + " ]\n", + " },\n", + " \"feature\": \"CLASSIFICATION\",\n", + " \"imageClassificationConfig\": {\n", + " \"annotationSpecSet\": \"projects/migration-ucaip-training/annotationSpecSets/60401d46_0000_2c8d_aa13_883d24f7e3c8\",\n", + " \"answerAggregationType\": \"MAJORITY_VOTE\"\n", + " }\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cNNI6bjcxXiY" + }, + "source": [ + "#### Call\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "request = clients[\"labeling\"].label_image(\n", + " request={\n", + " \"parent\": dataset_id,\n", + " \"basic_config\": basic_config,\n", + " \"feature\": feature,\n", + " \"image_classification_config\": config\n", + " }\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "UxcyIvGLxXiW" + }, + "source": [ + "#### Response\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cxbp1cGzxXiW" + }, + "source": [ + "*Example output*:\n", + "```\n", + "\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "labeling_job_name = request.operation.name\n", + "\n", + "print(labeling_job_name)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "yfufMwAEJjkX" + }, + "source": [ + "### [projects.operations.get](https://cloud.google.com/ai-platform/data-labeling/docs/reference/rest/v1beta1/projects.operations/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cNNI6bjcxXiY" + }, + "source": [ + "#### Call\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "request = clients[\"operations\"].get_operation(\n", + " name=labeling_job_name\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "UxcyIvGLxXiW" + }, + "source": [ + "#### Response\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cxbp1cGzxXiW" + }, + "source": [ + "*Example output*:\n", + "```\n", + "\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "yfufMwAEJjkX" + }, + "source": [ + "### [projects.operations.cancel](https://cloud.google.com/ai-platform/data-labeling/docs/reference/rest/v1beta1/projects.operations/cancel)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cNNI6bjcxXiY" + }, + "source": [ + "#### Call\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "request = clients[\"operations\"].cancel_operation(\n", + " name=labeling_job_name\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "UxcyIvGLxXiW" + }, + "source": [ + "#### Response\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cxbp1cGzxXiW" + }, + "source": [ + "*Example output*:\n", + "```\n", + "\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup:migration,new" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "IRwi0SHBxXio", + "scrolled": true + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_annotation_spec = True\n", + "delete_instruction = True\n", + "delete_labeling_job = True\n", + "delete_bucket = True\n", + "\n", + "\n", + "try:\n", + " if delete_dataset:\n", + " clients[\"operations\"].delete_operation(name=labeling_job_name)\n", + "except Exception as e:\n", + " print(e)\n", + " \n", + "try:\n", + " if delete_dataset:\n", + " clients[\"labeling\"].delete_instruction(name=instruction_name)\n", + "except Exception as e:\n", + " print(e)\n", + " \n", + "try:\n", + " if delete_dataset:\n", + " clients[\"labeling\"].delete_annotation_spec_set(name=annotation_spec_set_name)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "try:\n", + " if delete_dataset:\n", + " clients[\"labeling\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "\n", + "if delete_bucket and 'BUCKET_NAME' in globals():\n", + " ! gsutil rm -r gs://$BUCKET_NAME\n" + ] + } + ], + "metadata": { + "colab": { + "name": "[UJ.1 OLD] AutoML Vision Image Classification.ipynb", + "provenance": [] + }, + "environment": { + "name": "tf2-2-3-gpu.2-3.m55", + "type": "gcloud", + "uri": "gcr.io/deeplearning-platform-release/tf2-2-3-gpu.2-3:m55" + }, + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.7.8" + } + }, + "nbformat": 4, + "nbformat_minor": 4 +} diff --git a/notebooks/community/migration/UJ13 unified Data Labeling task.ipynb b/notebooks/community/migration/UJ13 unified Data Labeling task.ipynb new file mode 100644 index 000000000..68e18617f --- /dev/null +++ b/notebooks/community/migration/UJ13 unified Data Labeling task.ipynb @@ -0,0 +1,1463 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title:migration,new" + }, + "source": [ + "# Vertex SDK: Data Labeling\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest (preview) version of Vertex SDK.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "A_QVTkXr_i_r" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-aiplatform --user" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the Google *cloud-storage* library as well.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "65SBis8d_i_s" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-storage" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the Kernel\n", + "\n", + "Once you've installed the Vertex SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "GIwKc4pk_i_t" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "if not os.getenv(\"AUTORUN\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU run-time\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your GCP project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a GCP project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebooks.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex AI. We recommend when possible, to choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You cannot use a Multi-Regional Storage bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see [Region support for Vertex AI services](https://cloud.google.com/vertex-ai/docs/general/locations)\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "6Xtp5tvK_i_y" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "d70An-Mg_i_2" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your GCP account\n", + "\n", + "**If you are using Google Cloud Notebooks**, your environment is already\n", + "authenticated. Skip this step.\n", + "\n", + "*Note: If you are on an Vertex notebook and run the cell, the cell knows to skip executing the authentication steps.*\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "chybg3Ap_i_2" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your Google Cloud account. This provides access\n", + "# to your Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Vertex, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this tutorial in a notebook locally, replace the string\n", + " # below with the path to your service account key and run this cell to\n", + " # authenticate your Google Cloud account.\n", + " else:\n", + " %env GOOGLE_APPLICATION_CREDENTIALS your_path_to_credentials.json\n", + "\n", + " # Log in to your account on Google Cloud\n", + " ! gcloud auth login" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n", + " BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "qPPrwWpO_i_6" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "fn744B7x_i_7" + }, + "outputs": [], + "source": [ + "! gsutil ls -al gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip" + }, + "source": [ + "#### Import Vertex SDK\n", + "\n", + "Import the Vertex SDK into our Python environment.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "97-XQPkv_i_7" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex AI constants\n", + "\n", + "Setup up the following constants for Vertex AI:\n", + "\n", + "- `API_ENDPOINT`: The Vertex AI API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `API_PREDICT_ENDPOINT`: The Vertex AI API service endpoint for prediction.\n", + "- `PARENT`: The Vertex AI location root path for dataset, model and endpoint resources.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "kBBYqHEd_i_8" + }, + "outputs": [], + "source": [ + "# API Endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex AI location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants:automl" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Next, setup constants unique to AutoML image classification datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the managed dataset service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the managed dataset service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the managed pipelines service the task (e.g., classification) to train the model for.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:automl,icn" + }, + "outputs": [], + "source": [ + "# Image Dataset type\n", + "IMAGE_SCHEMA = \"google-cloud-aiplatform/schema/dataset/metadata/image_1.0.0.yaml\"\n", + "# Image Labeling type\n", + "IMPORT_SCHEMA_IMAGE_CLASSIFICATION = \"gs://google-cloud-aiplatform/schema/dataset/ioformat/image_classification_single_label_io_format_1.0.0.yaml\"\n", + "# Image labeling task\n", + "LABELING_SCHEMA_IMAGE = \"gs://google-cloud-aiplatform/schema/datalabelingjob/inputs/image_classification_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients" + }, + "source": [ + "## Clients\n", + "\n", + "The Vertex SDK works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the server (Vertex).\n", + "\n", + "You will use several clients in this tutorial, so set them all up upfront.\n", + "\n", + "- Dataset Service for managed datasets.\n", + "- Job Service for batch jobs and custom training.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "i2p2VYUz_i_-" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"job\"] = create_job_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:flowers,csv,icn" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "LABELING_FILES = [\n", + " \"https://raw.githubusercontent.com/googleapis/python-aiplatform/master/samples/snippets/resources/daisy.jpg\"\n", + "]\n", + "\n", + "IMPORT_FILE = \"gs://\" + BUCKET_NAME + \"/labeling.csv\"\n", + "with tf.io.gfile.GFile(IMPORT_FILE, \"w\") as f:\n", + " for lf in LABELING_FILES:\n", + " ! wget {lf} | gsutil cp {lf.split(\"/\")[-1]} gs://{BUCKET_NAME}\n", + " f.write(\"gs://\" + BUCKET_NAME + \"/\" + lf.split(\"/\")[-1] + \"\\n\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cBgFrDD1_i__" + }, + "outputs": [], + "source": [ + "! gsutil cat $IMPORT_FILE" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_create:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210303215432/daisy.jpg\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_a_dataset:migration" + }, + "source": [ + "## Create a dataset\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_create:migration,new" + }, + "source": [ + "### [projects.locations.datasets.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.datasets/create)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "request:migration" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "RsM4amS9_jAB" + }, + "outputs": [], + "source": [ + "DATA_SCHEMA = IMAGE_SCHEMA\n", + "\n", + "dataset = {\n", + " \"display_name\": \"labeling_\" + TIMESTAMP,\n", + " \"metadata_schema_uri\": \"gs://\" + DATA_SCHEMA,\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateDatasetRequest(parent=PARENT, dataset=dataset).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"dataset\": {\n", + " \"displayName\": \"labeling_20210303215432\",\n", + " \"metadataSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/metadata/image_1.0.0.yaml\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "call:migration" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "datasets_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "response:migration" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,response" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_create:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/datasets/1165112889535627264\",\n", + " \"displayName\": \"labeling_20210303215432\",\n", + " \"metadataSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/metadata/image_1.0.0.yaml\",\n", + " \"labels\": {\n", + " \"aiplatform.googleapis.com/dataset_metadata_schema\": \"IMAGE\"\n", + " },\n", + " \"metadata\": {\n", + " \"dataItemSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/dataitem/image_1.0.0.yaml\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new" + }, + "source": [ + "### [projects.locations.datasets.import](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.datasets/import)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "oJtL08_Q_jAF" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "outputs": [], + "source": [ + "LABEL_SCHEMA = IMPORT_SCHEMA_IMAGE_CLASSIFICATION\n", + "\n", + "import_config = {\n", + " \"gcs_source\": {\"uris\": [IMPORT_FILE]},\n", + " \"import_schema_uri\": LABEL_SCHEMA,\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.ImportDataRequest(\n", + " name=dataset_short_id, import_configs=[import_config]\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gC5MZDJc_jAG" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"1165112889535627264\",\n", + " \"importConfigs\": [\n", + " {\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210303215432/labeling.csv\"\n", + " ]\n", + " },\n", + " \"importSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/ioformat/image_classification_single_label_io_format_1.0.0.yaml\"\n", + " }\n", + " ]\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "rnhDF5vW_jAG" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "datasets_import:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"dataset\"].import_data(\n", + " name=dataset_id, import_configs=[import_config]\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "zQoFJ2K0_jAH" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "K7BjtQ3e_jAH" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "nkkcsM2Pq3jN" + }, + "source": [ + "### Create data labeling specialist pool\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wlZhYc8iq3jN" + }, + "source": [ + "In case you do not have access to labeling services execute this section." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ySEK40sHq3jN" + }, + "outputs": [], + "source": [ + "# add client for specialist pool\n", + "clients[\"specialist_pool\"] = aip.SpecialistPoolServiceClient(\n", + " client_options=client_options\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "trainingpipelines_create:migration,new" + }, + "source": [ + "### [projects.locations.specialistPools.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.specialistPools/createe)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cs4roPCHq3jO" + }, + "source": [ + "#### Request\n", + "\n", + "In this part, you will replace [your-email-address] with your email address. This makes you the specialist and recipient of the labeling request." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wsXOMVOMq3jO" + }, + "outputs": [], + "source": [ + "EMAIL = \"[your-email-address]\"\n", + "\n", + "specialist_pool = {\n", + " \"name\": \"labeling_\" + TIMESTAMP, # he resource name of the SpecialistPool.\n", + " \"display_name\": \"labeling_\" + TIMESTAMP, # user-defined name of the SpecialistPool\n", + " \"specialist_manager_emails\": [EMAIL],\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateSpecialistPoolRequest(\n", + " parent=PARENT, specialist_pool=specialist_pool\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bPT3DQwsq3jO" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"specialistPool\": {\n", + " \"name\": \"labeling_20210303215432\",\n", + " \"displayName\": \"labeling_20210303215432\",\n", + " \"specialistManagerEmails\": [\n", + " \"dev@fourteen33.com\"\n", + " ]\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ZGfGuq4gq3jO" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "LiuOtTM4q3jO" + }, + "outputs": [], + "source": [ + "request = clients[\"specialist_pool\"].create_specialist_pool(\n", + " parent=PARENT, specialist_pool=specialist_pool\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Ls4r2QDhq3jO" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "imFEkmH2q3jP" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "i1rCrqH2q3jP" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/specialistPools/1167839678372511744\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "lqPnPCUjq3jP" + }, + "outputs": [], + "source": [ + "specialist_name = result.name\n", + "\n", + "specialist_id = specialist_name.split(\"/\")[-1]\n", + "\n", + "print(specialist_name)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_a_model:migration" + }, + "source": [ + "## Create data labeling job\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "nsuFamYYq3jQ" + }, + "source": [ + "### [projects.locations.dataLabelingJobs.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.dataLabelingJobs/create)\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "5AY3SJjQq3jQ" + }, + "outputs": [], + "source": [ + "# create placeholder file for valid PDF file with instruction for data labeling\n", + "! echo \"this is instruction\" >> instruction.txt | gsutil cp instruction.txt gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "m5igPySU_jAJ" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_create:migration,new,request,icn" + }, + "outputs": [], + "source": [ + "LABLEING_SCHEMA = LABELING_SCHEMA_IMAGE\n", + "INSTRUCTION_FILE = \"gs://\" + BUCKET_NAME + \"/instruction.txt\"\n", + "\n", + "inputs = json_format.ParseDict({\"annotation_specs\": [\"rose\"]}, Value())\n", + "\n", + "data_labeling_job = {\n", + " \"display_name\": \"labeling_\" + TIMESTAMP,\n", + " \"datasets\": [dataset_id],\n", + " \"labeler_count\": 1,\n", + " \"instruction_uri\": INSTRUCTION_FILE,\n", + " \"inputs_schema_uri\": LABLEING_SCHEMA,\n", + " \"inputs\": inputs,\n", + " \"annotation_labels\": {\n", + " \"aiplatform.googleapis.com/annotation_set_name\": \"data_labeling_job_specialist_pool\"\n", + " },\n", + " \"specialist_pools\": [specialist_name],\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateDataLabelingJobRequest(\n", + " parent=PARENT, data_labeling_job=data_labeling_job\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "7IQ6Jp8E_jAJ" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"dataLabelingJob\": {\n", + " \"displayName\": \"labeling_20210303215432\",\n", + " \"datasets\": [\n", + " \"projects/116273516712/locations/us-central1/datasets/1165112889535627264\"\n", + " ],\n", + " \"labelerCount\": 1,\n", + " \"instructionUri\": \"gs://migration-ucaip-trainingaip-20210303215432/instruction.txt\",\n", + " \"inputsSchemaUri\": \"gs://google-cloud-aiplatform/schema/datalabelingjob/inputs/image_classification_1.0.0.yaml\",\n", + " \"inputs\": {\n", + " \"annotation_specs\": [\n", + " \"rose\"\n", + " ]\n", + " },\n", + " \"annotationLabels\": {\n", + " \"aiplatform.googleapis.com/annotation_set_name\": \"data_labeling_job_specialist_pool\"\n", + " },\n", + " \"specialistPools\": [\n", + " \"projects/116273516712/locations/us-central1/specialistPools/1167839678372511744\"\n", + " ]\n", + " }\n", + "\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "xhuR86RL_jAK" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].create_data_labeling_job(\n", + " parent=PARENT, data_labeling_job=data_labeling_job\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "S8zP7wju_jAL" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,request" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "trainingpipelines_create:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/dataLabelingJobs/3830883229125050368\",\n", + " \"displayName\": \"labeling_20210303215432\",\n", + " \"datasets\": [\n", + " \"projects/116273516712/locations/us-central1/datasets/1165112889535627264\"\n", + " ],\n", + " \"labelerCount\": 1,\n", + " \"instructionUri\": \"gs://migration-ucaip-trainingaip-20210303215432/instruction.txt\",\n", + " \"inputsSchemaUri\": \"gs://google-cloud-aiplatform/schema/datalabelingjob/inputs/image_classification_1.0.0.yaml\",\n", + " \"inputs\": {\n", + " \"annotationSpecs\": [\n", + " \"rose\"\n", + " ]\n", + " },\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"createTime\": \"2021-03-03T21:55:31.239049Z\",\n", + " \"updateTime\": \"2021-03-03T21:55:31.239049Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "training_pipeline_id:migration,new,response" + }, + "outputs": [], + "source": [ + "labeling_task_name = request.name\n", + "\n", + "print(labeling_task_name)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "trainingpipelines_get:migration,new" + }, + "source": [ + "### [projects.locations.dataLabelingJobs.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.dataLabelingJobs/get)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "A3jRv70o_jAN" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "rPnMOftyq3jS" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].get_data_labeling_job(name=labeling_task_name)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "XC5I2xxt_jAN" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "yXZnQR1t_jAO" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Df0XR2uMq3jS" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/dataLabelingJobs/3830883229125050368\",\n", + " \"displayName\": \"labeling_20210303215432\",\n", + " \"datasets\": [\n", + " \"projects/116273516712/locations/us-central1/datasets/1165112889535627264\"\n", + " ],\n", + " \"labelerCount\": 1,\n", + " \"instructionUri\": \"gs://migration-ucaip-trainingaip-20210303215432/instruction.txt\",\n", + " \"inputsSchemaUri\": \"gs://google-cloud-aiplatform/schema/datalabelingjob/inputs/image_classification_1.0.0.yaml\",\n", + " \"inputs\": {\n", + " \"annotationSpecs\": [\n", + " \"rose\"\n", + " ]\n", + " },\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"createTime\": \"2021-03-03T21:55:31.239049Z\",\n", + " \"updateTime\": \"2021-03-03T21:55:31.239049Z\",\n", + " \"specialistPools\": [\n", + " \"projects/116273516712/locations/us-central1/specialistPools/1167839678372511744\"\n", + " ]\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "fQ0TVlokq3jS" + }, + "source": [ + "### [projects.locations.dataLabelingJobs.cancel](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.dataLabelingJobs/cancel)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Mhfy-xKIq3jS" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "FZ3dlKJjq3jT" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].cancel_data_labeling_job(name=labeling_task_name)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ES_05aHoq3jT" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "jsIfaQ3Aq3jT" + }, + "outputs": [], + "source": [ + "print(request)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "LHnXRXrKq3jT" + }, + "source": [ + "*Example output*:\n", + "```\n", + "None\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = clients[\"job\"].get_data_labeling_job(name=labeling_task_name)\n", + " if response.state == aip.JobState.JOB_STATE_CANCELLED:\n", + " print(\"Labeling job CANCELED\")\n", + " break\n", + " else:\n", + " print(\"Canceling labeling job:\", response.state)\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup:migration,new" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aoJ18d8Y_jAy" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_job = True\n", + "delete_specialist_pool = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex AI fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset:\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the labeling job using the Vertex AI fully qualified identifier for the dataset\n", + "try:\n", + " if delete_job:\n", + " request = clients[\"job\"].delete_data_labeling_job(name=labeling_task_name)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the specialist pool using the Vertex AI fully qualified identifier for the dataset\n", + "try:\n", + " if delete_specialist_pool:\n", + " clients[\"specialist_pool\"].delete_specialist_pool(name=specialist_name)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r gs://$BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "collapsed_sections": [ + "timestamp", + "gcp_authenticate", + "bucket:batch_prediction", + "setup_vars", + "import_aip", + "aip_constants", + "automl_constants:automl", + "datasets_create:migration,new", + "request:migration", + "call:migration", + "response:migration", + "datasets_import:migration,new", + "oJtL08_Q_jAF", + "rnhDF5vW_jAG", + "zQoFJ2K0_jAH", + "trainingpipelines_create:migration,new", + "m5igPySU_jAJ", + "xhuR86RL_jAK", + "S8zP7wju_jAL", + "trainingpipelines_get:migration,new", + "A3jRv70o_jAN", + "XC5I2xxt_jAN", + "models_evaluations_list:migration,new", + "Ngn6qqVy_jAQ", + "F0ryqI3F_jAQ", + "models_evaluations_get:migration,new", + "_NXujm2U_jAR", + "0RLTdCfj_jAS", + "make_batch_prediction_file:migration,new", + "make_batch_file:automl,image", + "batchpredictionjobs_create:migration,new", + "htIpycBi_jAX", + "8QO3y-36_jAY", + "DmClxRYK_jAY", + "batchpredictionjobs_get:migration,new", + "aSE_wqES_jAa", + "LUy0NIF__jAa", + "endpoints_create:migration,new", + "Ph5S0j4v_jAc", + "yjsSo1cM_jAd", + "ijvF_HGd_jAe", + "endpoints_deploymodel:migration,new", + "NFIRI0XT_jAf", + "c3_4BVyW_jAh", + "7NmySa8R_jAh", + "endpoints_predict:migration,new", + "6fb84nKh_jAk", + "h6IskqWe_jAo", + "KHf2BSMR_jAo", + "endpoints_undeploymodel:migration,new", + "KrVZz6Uw_jAp", + "wvFK-kir_jAq", + "5bwEQMKT_jAr", + "LOsxiKj4_jAs", + "PWMEUCbF_jAt", + "OalQ6m9P_jAu", + "models_export:migration,new", + "lqJoqYMI_jAv", + "v6isqzPQ_jAw", + "ZCyd1qAb_jAx" + ], + "name": "UJ13 unified Data Labeling task.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/migration/UJ14 legacy AutoML Vision Video Classification.ipynb b/notebooks/community/migration/UJ14 legacy AutoML Vision Video Classification.ipynb new file mode 100644 index 000000000..5b63e1656 --- /dev/null +++ b/notebooks/community/migration/UJ14 legacy AutoML Vision Video Classification.ipynb @@ -0,0 +1,1554 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License.\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title:migration,new" + }, + "source": [ + "# AutoML SDK: AutoML image classification model\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest (preview) version of AutoML SDK.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "1pHEB-ZsMda_", + "outputId": "babb3ee0-14e0-43fc-afc7-7e6bd608bmigration" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-automl --user\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the Google *cloud-storage* library as well.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "QnnZhATdMdbB", + "outputId": "5c8e0911-d97b-4bec-bb6d-02f56441ac0d" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-storage\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the Kernel\n", + "\n", + "Once you've installed the AutoML SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "4pKvX2SwMdbD" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "\n", + "if not os.getenv(\"AUTORUN\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU run-time\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your GCP project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a GCP project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the AutoML APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [Google Cloud SDK](https://cloud.google.com/sdk) is already installed in AutoML Notebooks.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" #@param {type:\"string\"}\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id", + "outputId": "1433ef90-4d91-4070-9201-7c9071cb920f" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id", + "outputId": "66d8ef94-2134-4f39-95a9-917619534207" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for AutoML. We recommend when possible, to choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You cannot use a Multi-Regional Storage bucket for training with AutoML. Not all regions provide support for all AutoML services. For the latest support per region, see [Region support for AutoML services]()\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dZBY6GRCMdbI" + }, + "outputs": [], + "source": [ + "REGION = 'us-central1' #@param {type: \"string\"}\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wiCt8N48MdbJ" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your GCP account\n", + "\n", + "**If you are using AutoML Notebooks**, your environment is already\n", + "authenticated. Skip this step.\n", + "\n", + "*Note: If you are on an AutoML notebook and run the cell, the cell knows to skip executing the authentication steps.*\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "_nBq2y6zMdbK" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your Google Cloud account. This provides access\n", + "# to your Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on AutoML, then don't execute this code\n", + "if not os.path.exists('/opt/deeplearning/metadata/env_version'):\n", + " if 'google.colab' in sys.modules:\n", + " from google.colab import auth as google_auth\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this tutorial in a notebook locally, replace the string\n", + " # below with the path to your service account key and run this cell to\n", + " # authenticate your Google Cloud account.\n", + " else:\n", + " %env GOOGLE_APPLICATION_CREDENTIALS your_path_to_credentials.json\n", + "\n", + " # Log in to your account on Google Cloud\n", + " ! gcloud auth login\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"[your-bucket-name]\" #@param {type:\"string\"}\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n", + " BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Y5_NZRmyMdbP", + "outputId": "e5a3655d-8cf0-4cbd-9c18-f64bffb4ca00" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION gs://$BUCKET_NAME\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aRnzBOTfMdbQ" + }, + "outputs": [], + "source": [ + "! gsutil ls -al gs://$BUCKET_NAME\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip" + }, + "source": [ + "#### Import AutoML SDK\n", + "\n", + "Import the AutoML SDK into our Python environment.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "z_yz9CUZMdbR" + }, + "outputs": [], + "source": [ + "import json\n", + "import os\n", + "import sys\n", + "import time\n", + "\n", + "\n", + "from google.cloud import automl_v1beta1 as automl\n", + "\n", + "\n", + "from google.protobuf.json_format import MessageToJson\n", + "from google.protobuf.json_format import ParseDict\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Setup up the following constants for AutoML:\n", + "\n", + "- `PARENT`: The AutoML location root path for dataset, model and endpoint resources.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "_tmHDZHGMdbS" + }, + "outputs": [], + "source": [ + "# AutoML location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients" + }, + "source": [ + "## Clients\n", + "\n", + "The AutoML SDK works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the server (AutoML).\n", + "\n", + "You will use several clients in this tutorial, so set them all up upfront.\n", + "\n", + "(?)\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "OiFdpU4mMdbT", + "outputId": "d29dc331-d1f3-494d-ec73-e36176953a10" + }, + "outputs": [], + "source": [ + "def automl_client():\n", + " return automl.AutoMlClient()\n", + "\n", + "def prediction_client():\n", + " return automl.PredictionServiceClient()\n", + "\n", + "def operations_client():\n", + " return automl.AutoMlClient()._transport.operations_client\n", + "\n", + "clients = {}\n", + "clients[\"automl\"] = automl_client()\n", + "clients[\"prediction\"] = prediction_client()\n", + "clients[\"operations\"] = operations_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:flowers,csv,icn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = 'gs://automl-video-demo-data/hmdb_split1.csv'\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "KHJnNxI5xXiS", + "outputId": "7714d066-2301-4b47-d155-d3c5d454def7" + }, + "outputs": [], + "source": [ + "! gsutil cat $IMPORT_FILE | head -n 10 \n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_create:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "TRAIN,gs://automl-video-demo-data/hmdb_split1_5classes_train_inf.csv\n", + "TEST,gs://automl-video-demo-data/hmdb_split1_5classes_test_inf.csv\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_a_dataset:migration" + }, + "source": [ + "## Create a dataset\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "MuX78NX0MdbZ" + }, + "source": [ + "### [projects.locations.datasets.create](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.datasets/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "request:migration" + }, + "source": [ + "#### Request\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "nDWXlqoKcDnf", + "outputId": "bf41d872-b199-4272-b47b-2353b8c3b6cc" + }, + "outputs": [], + "source": [ + "dataset = {\n", + " \"display_name\": \"hmdb_\" + TIMESTAMP,\n", + " \"video_classification_dataset_metadata\": {}\n", + "}\n", + "\n", + "print(MessageToJson(\n", + " automl.CreateDatasetRequest(\n", + " parent=PARENT,\n", + " dataset=dataset\n", + " ).__dict__[\"_pb\"])\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pSSBWJ49Mdba" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"dataset\": {\n", + " \"displayName\": \"hmdb_20210228225744\",\n", + " \"videoClassificationDatasetMetadata\": {}\n", + " }\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "call:migration" + }, + "source": [ + "#### Call\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "datasets_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].create_dataset(\n", + " parent=PARENT,\n", + " dataset=dataset\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "response:migration" + }, + "source": [ + "#### Response\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,response", + "outputId": "99f328e4-877e-4d4c-d9a6-6f44a7f03a4c" + }, + "outputs": [], + "source": [ + "result = request\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_create:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/datasets/VCN6574174086275006464\",\n", + " \"displayName\": \"hmdb_20210228225744\",\n", + " \"createTime\": \"2021-02-28T23:06:43.197904Z\",\n", + " \"etag\": \"AB3BwFrtf0Yl4fgnXW4leoEEANTAGQdOngyIqdQSJBT9pKEChgeXom-0OyH7dKtfvA4=\",\n", + " \"videoClassificationDatasetMetadata\": {}\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:migration,new,response", + "outputId": "4af2c8c9-84c6-434b-a5c1-0bf199be69f5" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split('/')[-1]\n", + "\n", + "print(dataset_id)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cV7PMAk0Mdbi" + }, + "source": [ + "### [projects.locations.datasets.importData](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.datasets/importData)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "BFtdHDpfMdbi" + }, + "source": [ + "#### Request\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "datasets_import:migration,new,request", + "outputId": "8166381f-48f3-41e6-a86f-f101da50ceaa" + }, + "outputs": [], + "source": [ + "input_config = {\n", + " \"gcs_source\": {\n", + " \"input_uris\": [IMPORT_FILE]\n", + " }\n", + "}\n", + "\n", + "print(MessageToJson(\n", + " automl.ImportDataRequest(\n", + " name=dataset_short_id,\n", + " input_config=input_config\n", + " ).__dict__[\"_pb\"])\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "LzHxgr_6Mdbj" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"VCN6574174086275006464\",\n", + " \"inputConfig\": {\n", + " \"gcsSource\": {\n", + " \"inputUris\": [\n", + " \"gs://automl-video-demo-data/hmdb_split1.csv\"\n", + " ]\n", + " }\n", + " }\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "LWRK-vjJMdbk" + }, + "source": [ + "#### Call\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "datasets_import:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].import_data(\n", + " name=dataset_id,\n", + " input_config=input_config\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bUrXTi34Mdbl" + }, + "source": [ + "#### Response\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "fETkBwGaMdbl", + "outputId": "431361a6-8a4a-408b-b6f8-d631d2b6fe69" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_a_model:migration" + }, + "source": [ + "## Train a model\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_models_create:migration,old" + }, + "source": [ + "### [projects.locations.models.create](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "XhbhTx6DMdbo" + }, + "source": [ + "#### Request\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_create:migration,new,request,icn", + "outputId": "aaa36418-7aa8-4b6b-8b06-5343fa1b35e0" + }, + "outputs": [], + "source": [ + "model = {\n", + " \"display_name\": \"hmdb_\" + TIMESTAMP,\n", + " \"dataset_id\": dataset_short_id,\n", + " \"video_classification_model_metadata\": {}\n", + "}\n", + "\n", + "print(MessageToJson(\n", + " automl.CreateModelRequest(\n", + " parent=PARENT,\n", + " model=model\n", + " ).__dict__[\"_pb\"])\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "1__HJXz0Mdbq" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"model\": {\n", + " \"displayName\": \"hmdb_20210228225744\",\n", + " \"datasetId\": \"VCN6574174086275006464\",\n", + " \"videoClassificationModelMetadata\": {}\n", + " }\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "8_Z5loUyMdbq" + }, + "source": [ + "#### Call\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].create_model(\n", + " parent=PARENT,\n", + " model=model\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "fZLpwN_dMdbr" + }, + "source": [ + "#### Response\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,request", + "outputId": "7e7f19e1-771d-4e4e-8edc-87eab3672b9e" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "trainingpipelines_create:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/VCN6188818900239515648\"\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "training_pipeline_id:migration,new,response", + "outputId": "0de12c6d-ffac-470e-d39e-0d0b1a92f4c0" + }, + "outputs": [], + "source": [ + "# The full unique ID for the training pipeline\n", + "model_id = result.name\n", + "# The short numeric ID for the training pipeline\n", + "model_short_id = model_id.split('/')[-1]\n", + "\n", + "print(model_short_id)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:migration" + }, + "source": [ + "## Evaluate the model\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "yfufMwAEJjkX" + }, + "source": [ + "### [projects.locations.models.modelEvaluations.list](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models.modelEvaluations/list)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "x6tyDgg2Mdbt" + }, + "source": [ + "#### Call\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "models_evaluations_list:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].list_model_evaluations(\n", + " parent=model_id\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Pz4WmA2SMdbu" + }, + "source": [ + "#### Response\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "AxEY7WFHj2YC", + "outputId": "e284ae12-dbb8-4b8f-c576-03d23cc822af" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "\n", + "model_evaluations = [\n", + " json.loads(MessageToJson(me.__dict__[\"_pb\"])) for me in request \n", + "]\n", + "# The evaluation slice\n", + "evaluation_slice = request.model_evaluation[0].name\n", + "\n", + "print(json.dumps(model_evaluations, indent=2))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "models_evaluations_list:migration,new,response,icn" + }, + "source": [ + "*Example output*\n", + "```\n", + "[\n", + " {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/VCN6188818900239515648/modelEvaluations/1998146574672720266\",\n", + " \"createTime\": \"2021-03-01T01:02:02.452298Z\",\n", + " \"evaluatedExampleCount\": 150,\n", + " \"classificationEvaluationMetrics\": {\n", + " \"auPrc\": 1.0,\n", + " \"confidenceMetricsEntry\": [\n", + " {\n", + " \"confidenceThreshold\": 0.016075565,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.2,\n", + " \"f1Score\": 0.33333334\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.017114623,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.202977,\n", + " \"f1Score\": 0.3374578\n", + " },\n", + " \n", + " # REMOVED FOR BREVITY\n", + " \n", + " {\n", + " \"confidenceThreshold\": 0.9299338,\n", + " \"recall\": 0.033333335,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.06451613\n", + " }\n", + " ]\n", + " },\n", + " \"displayName\": \"golf\"\n", + " }\n", + "]\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "i6T0bzuNJjkY" + }, + "source": [ + "### [projects.locations.models.modelEvaluations.get](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models.modelEvaluations/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0FoWkw05Mdb0" + }, + "source": [ + "#### Call\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "models_evaluations_get:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].get_model_evaluation(\n", + " name=evaluation_slice\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "u2H0pOs9Mdb1" + }, + "source": [ + "#### Response\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "IlGAsJM2Mdb1", + "outputId": "14f4d517-f948-455b-8cfe-0658e4320c94", + "scrolled": true + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "models_evaluations_get:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/VCN6188818900239515648/modelEvaluations/1998146574672720266\",\n", + " \"createTime\": \"2021-03-01T01:02:02.452298Z\",\n", + " \"evaluatedExampleCount\": 150,\n", + " \"classificationEvaluationMetrics\": {\n", + " \"auPrc\": 1.0,\n", + " \"confidenceMetricsEntry\": [\n", + " {\n", + " \"confidenceThreshold\": 0.016075565,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.2,\n", + " \"f1Score\": 0.33333334\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.017114623,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.202977,\n", + " \"f1Score\": 0.3374578\n", + " },\n", + " \n", + " # REMOVED FOR BREVITY\n", + " \n", + " {\n", + " \"confidenceThreshold\": 0.9299338,\n", + " \"recall\": 0.006666667,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.013245033\n", + " }\n", + " ],\n", + " \"confusionMatrix\": {\n", + " \"annotationSpecId\": [\n", + " \"175274248095399936\",\n", + " \"2048771693081526272\",\n", + " \"4354614702295220224\",\n", + " \"6660457711508914176\",\n", + " \"8966300720722608128\"\n", + " ],\n", + " \"row\": [\n", + " {\n", + " \"exampleCount\": [\n", + " 30,\n", + " 0,\n", + " 0,\n", + " 0,\n", + " 0\n", + " ]\n", + " },\n", + " {\n", + " \"exampleCount\": [\n", + " 0,\n", + " 30,\n", + " 0,\n", + " 0,\n", + " 0\n", + " ]\n", + " },\n", + " {\n", + " \"exampleCount\": [\n", + " 0,\n", + " 0,\n", + " 30,\n", + " 0,\n", + " 0\n", + " ]\n", + " },\n", + " {\n", + " \"exampleCount\": [\n", + " 0,\n", + " 0,\n", + " 0,\n", + " 30,\n", + " 0\n", + " ]\n", + " },\n", + " {\n", + " \"exampleCount\": [\n", + " 0,\n", + " 0,\n", + " 0,\n", + " 0,\n", + " 30\n", + " ]\n", + " }\n", + " ],\n", + " \"displayName\": [\n", + " \"ride_horse\",\n", + " \"golf\",\n", + " \"cartwheel\",\n", + " \"pullup\",\n", + " \"kick_ball\"\n", + " ]\n", + " }\n", + " }\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_predictions:migration" + }, + "source": [ + "## Make batch predictions\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:automl,image" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "To request a batch of predictions from AutoML Video, create a CSV file that lists the Cloud Storage paths to the videos that you want to annotate. You can also specify a start and end time to tell AutoML Video to only annotate a segment (segment-level) of the video. The start time must be zero or greater and must be before the end time. The end time must be greater than the start time and less than or equal to the duration of the video. You can also use inf to indicate the end of a video.\n", + "\n", + "example: \n", + " `gs://my-videos-vcm/short_video_1.avi,0.0,5.566667` \n", + " `gs://my-videos-vcm/car_chase.avi,0.0,3.933333` \n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_items:automl,icn,csv", + "outputId": "5eed7f11-5c61-4fdc-9f65-96a31e4f1313" + }, + "outputs": [], + "source": [ + "TRAIN_FILES = \"gs://automl-video-demo-data/hmdb_split1_5classes_train_inf.csv\"\n", + "\n", + "test_items = ! gsutil cat $TRAIN_FILES | head -n2\n", + "\n", + "cols = str(test_items[0]).split(',')\n", + "test_item_1, test_label_1, test_start_1, test_end_1 = str(cols[0]), str(cols[1]), str(cols[2]), str(cols[3])\n", + "print(test_item_1, test_label_1)\n", + "\n", + "cols = str(test_items[1]).split(',')\n", + "test_item_2, test_label_2, test_start_2, test_end_2 = str(cols[0]), str(cols[1]), str(cols[2]), str(cols[3])\n", + "print(test_item_2, test_label_2)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aZYEuzIccDnq" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://automl-video-demo-data/hmdb51/_Rad_Schlag_die_Bank__cartwheel_f_cm_np1_le_med_0.avi cartwheel\n", + "gs://automl-video-demo-data/hmdb51/Acrobacias_de_un_fenomeno_cartwheel_f_cm_np1_ba_bad_8.avi cartwheel\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tK_A0CKQMdcF", + "outputId": "9effe058-4f8c-4312-9cd3-15869bd206c8" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "import json\n", + "\n", + "gcs_input_uri = \"gs://\" + BUCKET_NAME + '/test.csv'\n", + "with tf.io.gfile.GFile(gcs_input_uri, 'w') as f:\n", + " data = f\"{test_item_1}, {test_start_1}, {test_end_1}\"\n", + " f.write(data + '\\n')\n", + " data = f\"{test_item_2}, {test_start_2}, {test_end_2}\"\n", + " f.write(data + '\\n')\n", + " \n", + "print(gcs_input_uri)\n", + "! gsutil cat $gcs_input_uri\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "JjkG44EScDnq" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210228225744/test.csv\n", + "gs://automl-video-demo-data/hmdb51/_Rad_Schlag_die_Bank__cartwheel_f_cm_np1_le_med_0.avi, 0.0, inf\n", + "gs://automl-video-demo-data/hmdb51/Acrobacias_de_un_fenomeno_cartwheel_f_cm_np1_ba_bad_8.avi, 0.0, inf\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_models_batchpredict:migration,old" + }, + "source": [ + "### [projects.locations.models.batchPredict](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models/batchPredict)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "YvQJiFamMdcG" + }, + "source": [ + "#### Request\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_create:migration,new,request,icn", + "outputId": "6ffce530-ea59-4b6b-c990-5cd684cc4c22" + }, + "outputs": [], + "source": [ + "input_config = {\n", + " \"gcs_source\": {\n", + " \"input_uris\": [gcs_input_uri]\n", + " }\n", + "}\n", + "\n", + "output_config = {\n", + " \"gcs_destination\": {\n", + " \"output_uri_prefix\": \"gs://\" + f\"{BUCKET_NAME}/batch_output/\"\n", + " }\n", + "}\n", + "\n", + "batch_prediction = automl.BatchPredictRequest(\n", + " name=model_id,\n", + " input_config=input_config,\n", + " output_config=output_config\n", + ")\n", + " \n", + "print(MessageToJson(\n", + " batch_prediction.__dict__[\"_pb\"])\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "zlU4Mbd5MdcJ" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/VCN6188818900239515648\",\n", + " \"inputConfig\": {\n", + " \"gcsSource\": {\n", + " \"inputUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210228225744/test.csv\"\n", + " ]\n", + " }\n", + " },\n", + " \"outputConfig\": {\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210228225744/batch_output/\"\n", + " }\n", + " }\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "qjugtUM7MdcJ" + }, + "source": [ + "#### Call\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"prediction\"].batch_predict(\n", + " request=batch_prediction\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "u1lEkLr8MdcK" + }, + "source": [ + "#### Response\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Q6HGGVlaMdcK", + "outputId": "160c9c08-f46a-4e07-cb7e-b558ddd23a39" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_create:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "\n", + "```\n", + "{}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup:migration,new" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wuJTfRWNMdcL", + "outputId": "4e7f019b-c477-42db-ae3f-4a3a78e1df88" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_model = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the AutoML fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset:\n", + " clients['automl'].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the AutoML fully qualified identifier for the model\n", + "try:\n", + " if delete_model:\n", + " clients['automl'].delete_model(name=model_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and 'BUCKET_NAME' in globals():\n", + " ! gsutil rm -r gs://$BUCKET_NAME\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Z5i6Q68BcDns" + }, + "outputs": [], + "source": [] + } + ], + "metadata": { + "colab": { + "name": "[UJ.14 OLD] AutoML Vision Video Classification.ipynb", + "provenance": [], + "toc_visible": true + }, + "environment": { + "name": "tf2-2-3-gpu.2-3.m55", + "type": "gcloud", + "uri": "gcr.io/deeplearning-platform-release/tf2-2-3-gpu.2-3:m55" + }, + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.7.8" + } + }, + "nbformat": 4, + "nbformat_minor": 4 +} diff --git a/notebooks/community/migration/UJ14 unified AutoML Vision Video Classification.ipynb b/notebooks/community/migration/UJ14 unified AutoML Vision Video Classification.ipynb new file mode 100644 index 000000000..a6dcb66b5 --- /dev/null +++ b/notebooks/community/migration/UJ14 unified AutoML Vision Video Classification.ipynb @@ -0,0 +1,1996 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title:migration,new" + }, + "source": [ + "# Vertex SDK: AutoML video classification model\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest (preview) version of Vertex SDK.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "M8_SmXFMSnkR" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-aiplatform --user" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the Google *cloud-storage* library as well.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "NJ11e4b0SnkS" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-storage" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the Kernel\n", + "\n", + "Once you've installed the Vertex SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ZL_bvLeVSnkT" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "if not os.getenv(\"AUTORUN\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU run-time\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your GCP project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a GCP project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebooks.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex AI. We recommend when possible, to choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You cannot use a Multi-Regional Storage bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see [Region support for Vertex AI services](https://cloud.google.com/vertex-ai/docs/general/locations)\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pHZF5f0tSnkV" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "xlsBZmnFSnkV" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your GCP account\n", + "\n", + "**If you are using Google Cloud Notebooks**, your environment is already\n", + "authenticated. Skip this step.\n", + "\n", + "*Note: If you are on an Vertex notebook and run the cell, the cell knows to skip executing the authentication steps.*\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "z78YvQXKSnkW" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your Google Cloud account. This provides access\n", + "# to your Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Vertex, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this tutorial in a notebook locally, replace the string\n", + " # below with the path to your service account key and run this cell to\n", + " # authenticate your Google Cloud account.\n", + " else:\n", + " %env GOOGLE_APPLICATION_CREDENTIALS your_path_to_credentials.json\n", + "\n", + " # Log in to your account on Google Cloud\n", + " ! gcloud auth login" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n", + " BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "SZHBMhUgSnkX" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "7GarvfseSnkY" + }, + "outputs": [], + "source": [ + "! gsutil ls -al gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip" + }, + "source": [ + "#### Import Vertex SDK\n", + "\n", + "Import the Vertex SDK into our Python environment.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "oAcxJMSBSnka" + }, + "outputs": [], + "source": [ + "import json\n", + "import os\n", + "import sys\n", + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex AI constants\n", + "\n", + "Setup up the following constants for Vertex AI:\n", + "\n", + "- `API_ENDPOINT`: The Vertex AI API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex AI location root path for dataset, model and endpoint resources.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "U0KV2IX7Snka" + }, + "outputs": [], + "source": [ + "# API Endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex AI location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants:automl" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Next, setup constants unique to AutoML video classification datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the managed dataset service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the managed dataset service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the managed pipelines service the task (e.g., classification) to train the model for.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:automl,icn" + }, + "outputs": [], + "source": [ + "# Video Dataset type\n", + "VIDEO_SCHEMA = \"google-cloud-aiplatform/schema/dataset/metadata/video_1.0.0.yaml\"\n", + "# Video Labeling type\n", + "IMPORT_SCHEMA_VIDEO_CLASSIFICATION = \"gs://google-cloud-aiplatform/schema/dataset/ioformat/video_classification_io_format_1.0.0.yaml\"\n", + "# Video Training task\n", + "TRAINING_VIDEO_CLASSIFICATION_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_video_classification_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients" + }, + "source": [ + "## Clients\n", + "\n", + "The Vertex SDK works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the server (Vertex).\n", + "\n", + "You will use several clients in this tutorial, so set them all up upfront.\n", + "\n", + "- Dataset Service for managed datasets.\n", + "- Model Service for managed models.\n", + "- Pipeline Service for training.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving. *Note*: Prediction has a different service endpoint.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "PzGAytHrSnkb" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "clients[\"job\"] = create_job_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:flowers,csv,icn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://automl-video-demo-data/hmdb_split1_5classes_train_inf.csv\"" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Y7gEKRMpSnkc" + }, + "outputs": [], + "source": [ + "! gsutil cat $IMPORT_FILE | head -n 10" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_create:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://automl-video-demo-data/hmdb51/_Rad_Schlag_die_Bank__cartwheel_f_cm_np1_le_med_0.avi,cartwheel,0.0,inf\n", + "gs://automl-video-demo-data/hmdb51/Acrobacias_de_un_fenomeno_cartwheel_f_cm_np1_ba_bad_8.avi,cartwheel,0.0,inf\n", + "gs://automl-video-demo-data/hmdb51/Acrobacias_de_un_fenomeno_cartwheel_f_cm_np1_fr_bad_3.avi,cartwheel,0.0,inf\n", + "gs://automl-video-demo-data/hmdb51/Acrobacias_de_un_fenomeno_cartwheel_f_cm_np1_fr_bad_4.avi,cartwheel,0.0,inf\n", + "gs://automl-video-demo-data/hmdb51/Acrobacias_de_un_fenomeno_cartwheel_f_cm_np1_fr_bad_5.avi,cartwheel,0.0,inf\n", + "gs://automl-video-demo-data/hmdb51/Bayer__Meisterin_Teresa_Stadler_cartwheel_f_cm_np1_le_med_0.avi,cartwheel,0.0,inf\n", + "gs://automl-video-demo-data/hmdb51/Bayer__Meisterin_Teresa_Stadler_cartwheel_f_cm_np1_le_med_2.avi,cartwheel,0.0,inf\n", + "gs://automl-video-demo-data/hmdb51/Boden_bung_Spoho_Eignungspr_fung_cartwheel_f_cm_np1_ri_med_2.avi,cartwheel,0.0,inf\n", + "gs://automl-video-demo-data/hmdb51/Bodenturnen_2004_cartwheel_f_cm_np1_le_med_0.avi,cartwheel,0.0,inf\n", + "gs://automl-video-demo-data/hmdb51/Bodenturnen_2004_cartwheel_f_cm_np1_le_med_4.avi,cartwheel,0.0,inf\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_a_dataset:migration" + }, + "source": [ + "## Create a dataset\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_create:migration,new" + }, + "source": [ + "### [projects.locations.datasets.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.datasets/create)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "request:migration" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "49ZUBDbWSnkd" + }, + "outputs": [], + "source": [ + "DATA_SCHEMA = VIDEO_SCHEMA\n", + "\n", + "dataset = {\n", + " \"display_name\": \"hmdb_\" + TIMESTAMP,\n", + " \"metadata_schema_uri\": \"gs://\" + DATA_SCHEMA,\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateDatasetRequest(parent=PARENT, dataset=dataset).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "W385QAT-Snke" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"dataset\": {\n", + " \"displayName\": \"hmdb_20210228191029\",\n", + " \"metadataSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/metadata/video_1.0.0.yaml\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "call:migration" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "datasets_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "response:migration" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,response" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_create:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/datasets/7952037527982964736\",\n", + " \"displayName\": \"hmdb_20210228191029\",\n", + " \"metadataSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/metadata/video_1.0.0.yaml\",\n", + " \"labels\": {\n", + " \"aiplatform.googleapis.com/dataset_metadata_schema\": \"VIDEO\"\n", + " },\n", + " \"metadata\": {\n", + " \"dataItemSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/dataitem/video_1.0.0.yaml\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)\n", + "\n", + "# Saved for clean up\n", + "dataset = {\"name\": dataset_id}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new" + }, + "source": [ + "### [projects.locations.datasets.import](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.datasets/import)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "MIkOh3jlSnkf" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "outputs": [], + "source": [ + "LABEL_SCHEMA = IMPORT_SCHEMA_VIDEO_CLASSIFICATION\n", + "\n", + "import_config = {\n", + " \"gcs_source\": {\"uris\": [IMPORT_FILE]},\n", + " \"import_schema_uri\": LABEL_SCHEMA,\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.ImportDataRequest(\n", + " name=dataset_short_id, import_configs=[import_config]\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "QEBKC5l-Snkf" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"7952037527982964736\",\n", + " \"importConfigs\": [\n", + " {\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://automl-video-demo-data/hmdb_split1_5classes_train_inf.csv\"\n", + " ]\n", + " },\n", + " \"importSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/ioformat/video_classification_io_format_1.0.0.yaml\"\n", + " }\n", + " ]\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gOWz5vTMSnkg" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "datasets_import:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"dataset\"].import_data(\n", + " name=dataset_id, import_configs=[import_config]\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pDdBxlyvSnkg" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "rF7pjlOYSnkg" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_a_model:migration" + }, + "source": [ + "## Train a model\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "trainingpipelines_create:migration,new" + }, + "source": [ + "### [projects.locations.trainingPipelines.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.trainingPipelines/create)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "J06YDrKMSnkh" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_create:migration,new,request,icn" + }, + "outputs": [], + "source": [ + "TRAINING_SCHEMA = TRAINING_VIDEO_CLASSIFICATION_SCHEMA\n", + "\n", + "task = ParseDict({}, Value())\n", + "\n", + "training_pipeline = {\n", + " \"display_name\": \"hmdb_\" + TIMESTAMP,\n", + " \"input_data_config\": {\"dataset_id\": dataset_short_id},\n", + " \"model_to_upload\": {\"display_name\": \"hmdb_\" + TIMESTAMP},\n", + " \"training_task_definition\": TRAINING_SCHEMA,\n", + " \"training_task_inputs\": task,\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateTrainingPipelineRequest(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "jPuxBFeaSnkh" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"trainingPipeline\": {\n", + " \"displayName\": \"hmdb_20210228191029\",\n", + " \"inputDataConfig\": {\n", + " \"datasetId\": \"7952037527982964736\"\n", + " },\n", + " \"trainingTaskDefinition\": \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_video_classification_1.0.0.yaml\",\n", + " \"trainingTaskInputs\": {},\n", + " \"modelToUpload\": {\n", + " \"displayName\": \"hmdb_20210228191029\"\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "AswA1FGoSnkk" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "orNAOfwiSnkk" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,request" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "trainingpipelines_create:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/trainingPipelines/3361945917925097472\",\n", + " \"displayName\": \"hmdb_20210228191029\",\n", + " \"inputDataConfig\": {\n", + " \"datasetId\": \"7952037527982964736\"\n", + " },\n", + " \"trainingTaskDefinition\": \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_video_classification_1.0.0.yaml\",\n", + " \"trainingTaskInputs\": {},\n", + " \"modelToUpload\": {\n", + " \"displayName\": \"hmdb_20210228191029\"\n", + " },\n", + " \"state\": \"PIPELINE_STATE_PENDING\",\n", + " \"createTime\": \"2021-02-28T19:17:39.815377Z\",\n", + " \"updateTime\": \"2021-02-28T19:17:39.815377Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "training_pipeline_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the training pipeline\n", + "training_pipeline_id = request.name\n", + "# The short numeric ID for the training pipeline\n", + "training_pipeline_short_id = training_pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(training_pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "trainingpipelines_get:migration,new" + }, + "source": [ + "### [projects.locations.trainingPipelines.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.trainingPipelines/get)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "vWsdsTkRSnkm" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_get:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"pipeline\"].get_training_pipeline(name=training_pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "HKb79nnsSnkm" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "UFnpgaznSnkm" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "trainingpipelines_get:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/trainingPipelines/3361945917925097472\",\n", + " \"displayName\": \"hmdb_20210228191029\",\n", + " \"inputDataConfig\": {\n", + " \"datasetId\": \"7952037527982964736\"\n", + " },\n", + " \"trainingTaskDefinition\": \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_video_classification_1.0.0.yaml\",\n", + " \"trainingTaskInputs\": {},\n", + " \"modelToUpload\": {\n", + " \"displayName\": \"hmdb_20210228191029\"\n", + " },\n", + " \"state\": \"PIPELINE_STATE_RUNNING\",\n", + " \"createTime\": \"2021-02-28T19:17:39.815377Z\",\n", + " \"startTime\": \"2021-02-28T19:17:40.089331Z\",\n", + " \"updateTime\": \"2021-02-28T19:17:40.089331Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=training_pipeline_id)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_name = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " break\n", + " else:\n", + " model_id = response.model_to_upload.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(20)\n", + "\n", + "print(model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:migration" + }, + "source": [ + "## Evaluate the model\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "models_evaluations_list:migration,new" + }, + "source": [ + "### [projects.locations.models.evaluations.list](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.models.evaluations/list)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "BmzDR40DSnko" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "models_evaluations_list:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"model\"].list_model_evaluations(parent=model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "BnR_19BsSnko" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "models_evaluations_list:migration,new,response" + }, + "outputs": [], + "source": [ + "model_evaluations = [json.loads(MessageToJson(mel.__dict__[\"_pb\"])) for mel in request]\n", + "\n", + "print(json.dumps(model_evaluations, indent=2))\n", + "\n", + "# The evaluation slice\n", + "evaluation_slice = request.model_evaluations[0].name" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "models_evaluations_list:migration,new,response,icn" + }, + "source": [ + "*Example output*\n", + "```\n", + "[\n", + " {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/5031242063400665088/evaluations/6719412425478635520\",\n", + " \"metricsSchemaUri\": \"gs://google-cloud-aiplatform/schema/modelevaluation/classification_metrics_1.0.0.yaml\",\n", + " \"metrics\": {\n", + " \"confidenceMetrics\": [\n", + " {\n", + " \"confidenceThreshold\": 0.0891612,\n", + " \"precision\": 0.2,\n", + " \"recall\": 1.0,\n", + " \"f1Score\": 0.33333334\n", + " },\n", + " {\n", + " \"recall\": 1.0,\n", + " \"confidenceThreshold\": 0.09073429,\n", + " \"precision\": 0.20289855,\n", + " \"f1Score\": 0.33734939\n", + " },\n", + " {\n", + " \"recall\": 1.0,\n", + " \"f1Score\": 0.34146342,\n", + " \"confidenceThreshold\": 0.09176466,\n", + " \"precision\": 0.20588236\n", + " },\n", + " \n", + " # REMOVED FOR BREVITY\n", + " \n", + " {\n", + " {\n", + " \"displayName\": \"pullup\",\n", + " \"id\": \"2856417959264387072\"\n", + " },\n", + " {\n", + " \"displayName\": \"golf\",\n", + " \"id\": \"5162260968478081024\"\n", + " },\n", + " {\n", + " \"displayName\": \"ride_horse\",\n", + " \"id\": \"6315182473084928000\"\n", + " },\n", + " {\n", + " \"displayName\": \"cartwheel\",\n", + " \"id\": \"7468103977691774976\"\n", + " }\n", + " ]\n", + " }\n", + " },\n", + " \"createTime\": \"2021-02-28T20:56:43.050002Z\",\n", + " \"sliceDimensions\": [\n", + " \"annotationSpec\"\n", + " ]\n", + " }\n", + "]\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "models_evaluations_get:migration,new" + }, + "source": [ + "### [projects.locations.models.evaluations.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.models.evaluations/get)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "1Zd-nTWkSnkp" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "models_evaluations_get:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"model\"].get_model_evaluation(name=evaluation_slice)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "XsNAiiycSnkp" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "iAibbLslSnkq" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "models_evaluations_get:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/5031242063400665088/evaluations/6719412425478635520\",\n", + " \"metricsSchemaUri\": \"gs://google-cloud-aiplatform/schema/modelevaluation/classification_metrics_1.0.0.yaml\",\n", + " \"metrics\": {\n", + " \"confusionMatrix\": {\n", + " \"rows\": [\n", + " [\n", + " 14.0,\n", + " 0.0,\n", + " 0.0,\n", + " 0.0,\n", + " 0.0\n", + " ],\n", + " [\n", + " 0.0,\n", + " 14.0,\n", + " 0.0,\n", + " 0.0,\n", + " 0.0\n", + " ],\n", + " [\n", + " 0.0,\n", + " 0.0,\n", + " 14.0,\n", + " 0.0,\n", + " 0.0\n", + " ],\n", + " [\n", + " 0.0,\n", + " 0.0,\n", + " 0.0,\n", + " 14.0,\n", + " 0.0\n", + " ],\n", + " [\n", + " 0.0,\n", + " 0.0,\n", + " 0.0,\n", + " 0.0,\n", + " 14.0\n", + " ]\n", + " ],\n", + " \"annotationSpecs\": [\n", + " {\n", + " \"displayName\": \"kick_ball\",\n", + " \"id\": \"1703496454657540096\"\n", + " },\n", + " {\n", + " \"displayName\": \"pullup\",\n", + " \"id\": \"2856417959264387072\"\n", + " },\n", + " {\n", + " \"displayName\": \"golf\",\n", + " \"id\": \"5162260968478081024\"\n", + " },\n", + " {\n", + " \"displayName\": \"ride_horse\",\n", + " \"id\": \"6315182473084928000\"\n", + " },\n", + " {\n", + " \"displayName\": \"cartwheel\",\n", + " \"id\": \"7468103977691774976\"\n", + " }\n", + " ]\n", + " },\n", + " \"confidenceMetrics\": [\n", + " {\n", + " \"precision\": 0.2,\n", + " \"recall\": 1.0,\n", + " \"confidenceThreshold\": 0.0891612,\n", + " \"f1Score\": 0.33333334\n", + " },\n", + " {\n", + " \"recall\": 1.0,\n", + " \"f1Score\": 0.33734939,\n", + " \"confidenceThreshold\": 0.09073429,\n", + " \"precision\": 0.20289855\n", + " },\n", + " {\n", + " \"precision\": 0.20588236,\n", + " \"f1Score\": 0.34146342,\n", + " \"confidenceThreshold\": 0.09176466,\n", + " \"recall\": 1.0\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.09279072,\n", + " \"f1Score\": 0.34739456,\n", + " \"precision\": 0.2102102,\n", + " \"recall\": 1.0\n", + " },\n", + " \n", + " # REMOVED FOR BREVITY\n", + " \n", + " {\n", + " \"recall\": 0.071428575,\n", + " \"f1Score\": 0.13333334,\n", + " \"precision\": 1.0,\n", + " \"confidenceThreshold\": 0.6023364\n", + " },\n", + " {\n", + " \"f1Score\": 0.055555556,\n", + " \"precision\": 1.0,\n", + " \"confidenceThreshold\": 0.6101756,\n", + " \"recall\": 0.028571429\n", + " },\n", + " {\n", + " \"recall\": 0.014285714,\n", + " \"precision\": 1.0,\n", + " \"confidenceThreshold\": 0.6113689,\n", + " \"f1Score\": 0.028169014\n", + " }\n", + " ],\n", + " \"auPrc\": 1.0\n", + " },\n", + " \"createTime\": \"2021-02-28T20:56:43.050002Z\",\n", + " \"sliceDimensions\": [\n", + " \"annotationSpec\"\n", + " ]\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_predictions:migration" + }, + "source": [ + "## Make batch predictions\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_prediction_file:migration,new" + }, + "source": [ + "### Make a batch prediction file\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_items:automl,icn,csv" + }, + "outputs": [], + "source": [ + "test_items = ! gsutil cat $IMPORT_FILE | head -n2\n", + "\n", + "cols = str(test_items[0]).split(\",\")\n", + "test_item_1 = str(cols[0])\n", + "test_label_1 = str(cols[1])\n", + "\n", + "print(test_item_1, test_label_1)\n", + "\n", + "cols = str(test_items[1]).split(\",\")\n", + "test_item_2 = str(cols[0])\n", + "test_label_2 = str(cols[1])\n", + "\n", + "print(test_item_2, test_label_2)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "kWOTd72cSnkr" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://automl-video-demo-data/hmdb51/_Rad_Schlag_die_Bank__cartwheel_f_cm_np1_le_med_0.avi cartwheel\n", + "gs://automl-video-demo-data/hmdb51/Acrobacias_de_un_fenomeno_cartwheel_f_cm_np1_ba_bad_8.avi cartwheel\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:automl,image" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "Let's now make a batch input file, which you store in your local Cloud Storage bucket. The batch input file can be either CSV or JSONL. You will use JSONL in this tutorial. For JSONL file, you make one dictionary entry per line for each video. The dictionary contains the key/value pairs:\n", + "\n", + "- `content`: The Cloud Storage path to the video.\n", + "- `mimeType`: The content type. In our example, it is an `avi` file.\n", + "- `timeSegmentStart`: The start timestamp in the video to do prediction on. *Note*, the timestamp must be specified as a string and followed by s (second), m (minute) or h (hour).\n", + "- `timeSegmentEnd`: The end timestamp in the video to do prediction on.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "yqU4Vz8TSnkr" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "import tensorflow as tf\n", + "\n", + "gcs_input_uri = \"gs://\" + BUCKET_NAME + \"/test.jsonl\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " data = {\n", + " \"content\": test_item_1,\n", + " \"mimeType\": \"video/avi\",\n", + " \"timeSegmentStart\": \"0.0s\",\n", + " \"timeSegmentEnd\": \"inf\",\n", + " }\n", + " f.write(json.dumps(data) + \"\\n\")\n", + " data = {\n", + " \"content\": test_item_2,\n", + " \"mimeType\": \"video/avi\",\n", + " \"timeSegmentStart\": \"0.0s\",\n", + " \"timeSegmentEnd\": \"inf\",\n", + " }\n", + " f.write(json.dumps(data) + \"\\n\")\n", + "\n", + "print(gcs_input_uri)\n", + "\n", + "!gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "OgdqIei1Snks" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210228191029/test.jsonl\n", + "{\"content\": \"gs://automl-video-demo-data/hmdb51/_Rad_Schlag_die_Bank__cartwheel_f_cm_np1_le_med_0.avi\", \"mimeType\": \"video/avi\", \"timeSegmentStart\": \"0.0s\", \"timeSegmentEnd\": \"inf\"}\n", + "{\"content\": \"gs://automl-video-demo-data/hmdb51/Acrobacias_de_un_fenomeno_cartwheel_f_cm_np1_ba_bad_8.avi\", \"mimeType\": \"video/avi\", \"timeSegmentStart\": \"0.0s\", \"timeSegmentEnd\": \"inf\"}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_create:migration,new" + }, + "source": [ + "### [projects.locations.batchPredictionJobs.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.batchPredictionJobs/create)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "hakbUKF5Snks" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_create:migration,new,request,icn" + }, + "outputs": [], + "source": [ + "batch_prediction_job = {\n", + " \"display_name\": \"hmdb_\" + TIMESTAMP,\n", + " \"model\": model_id,\n", + " \"input_config\": {\n", + " \"instances_format\": \"jsonl\",\n", + " \"gcs_source\": {\"uris\": [gcs_input_uri]},\n", + " },\n", + " \"model_parameters\": ParseDict(\n", + " {\n", + " \"confidenceThreshold\": 0.5,\n", + " \"maxPredictions\": 2,\n", + " \"segmentClassification\": True,\n", + " \"shotClassification\": True,\n", + " \"oneSecIntervalClassification\": True,\n", + " },\n", + " Value(),\n", + " ),\n", + " \"output_config\": {\n", + " \"predictions_format\": \"jsonl\",\n", + " \"gcs_destination\": {\n", + " \"output_uri_prefix\": \"gs://\" + f\"{BUCKET_NAME}/batch_output/\"\n", + " },\n", + " },\n", + " \"dedicated_resources\": {\n", + " \"machine_spec\": {\"machine_type\": \"n1-standard-2\", \"accelerator_count\": 0},\n", + " \"starting_replica_count\": 1,\n", + " \"max_replica_count\": 1,\n", + " },\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateBatchPredictionJobRequest(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "p0noQCPsSnkt" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"batchPredictionJob\": {\n", + " \"displayName\": \"hmdb_20210228191029\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/5031242063400665088\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210228191029/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"modelParameters\": {\n", + " \"segmentClassification\": true,\n", + " \"maxPredictions\": 2.0,\n", + " \"shotClassification\": true,\n", + " \"confidenceThreshold\": 0.5,\n", + " \"oneSecIntervalClassification\": true\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210228191029/batch_output/\"\n", + " }\n", + " },\n", + " \"dedicatedResources\": {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-2\"\n", + " },\n", + " \"startingReplicaCount\": 1,\n", + " \"maxReplicaCount\": 1\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "GQtFoN4eSnkt" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].create_batch_prediction_job(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "OY2yKVV_Snku" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Iu7YyWWgSnku" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_create:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/batchPredictionJobs/5275975759557558272\",\n", + " \"displayName\": \"hmdb_20210228191029\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/5031242063400665088\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210228191029/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"modelParameters\": {\n", + " \"oneSecIntervalClassification\": true,\n", + " \"confidenceThreshold\": 0.5,\n", + " \"maxPredictions\": 2.0,\n", + " \"shotClassification\": true,\n", + " \"segmentClassification\": true\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210228191029/batch_output/\"\n", + " }\n", + " },\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"completionStats\": {\n", + " \"incompleteCount\": \"-1\"\n", + " },\n", + " \"createTime\": \"2021-02-28T22:47:21.875565Z\",\n", + " \"updateTime\": \"2021-02-28T22:47:21.875565Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batch_job_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The fully qualified ID for the batch job\n", + "batch_job_id = request.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split(\"/\")[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_get:migration,new" + }, + "source": [ + "### [projects.locations.batchPredictionJobs.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.batchPredictionJobs/get)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "7RdKtPhlSnkw" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_get:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].get_batch_prediction_job(name=batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Hgt8RcLLSnkw" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "jhu1npewSnkw" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_get:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/batchPredictionJobs/5275975759557558272\",\n", + " \"displayName\": \"hmdb_20210228191029\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/5031242063400665088\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210228191029/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"modelParameters\": {\n", + " \"oneSecIntervalClassification\": true,\n", + " \"confidenceThreshold\": 0.5,\n", + " \"shotClassification\": true,\n", + " \"maxPredictions\": 2.0,\n", + " \"segmentClassification\": true\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210228191029/batch_output/\"\n", + " }\n", + " },\n", + " \"state\": \"JOB_STATE_RUNNING\",\n", + " \"completionStats\": {\n", + " \"incompleteCount\": \"2\"\n", + " },\n", + " \"createTime\": \"2021-02-28T22:47:21.875565Z\",\n", + " \"startTime\": \"2021-02-28T22:47:22.041508Z\",\n", + " \"updateTime\": \"2021-02-28T22:47:22.486289Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "def get_latest_predictions(gcs_out_dir):\n", + " \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n", + " folders = !gsutil ls $gcs_out_dir\n", + " latest = \"\"\n", + " for folder in folders:\n", + " subfolder = folder.split(\"/\")[-2]\n", + " if subfolder.startswith(\"prediction-\"):\n", + " if subfolder > latest:\n", + " latest = folder[:-1]\n", + " return latest\n", + "\n", + "\n", + "while True:\n", + " response = clients[\"job\"].get_batch_prediction_job(name=batch_job_id)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", response.state)\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " folder = get_latest_predictions(\n", + " response.output_config.gcs_destination.output_uri_prefix\n", + " )\n", + " ! gsutil ls $folder/prediction*.jsonl\n", + "\n", + " ! gsutil cat $folder/prediction*.jsonl\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_get:migration,new,wait,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210228191029/batch_output/prediction-hmdb_20210228191029-2021-02-28T22:47:21.701608Z/predictions_00001.jsonl\n", + "gs://migration-ucaip-trainingaip-20210228191029/batch_output/prediction-hmdb_20210228191029-2021-02-28T22:47:21.701608Z/predictions_00002.jsonl\n", + "{\"instance\":{\"content\":\"gs://automl-video-demo-data/hmdb51/Acrobacias_de_un_fenomeno_cartwheel_f_cm_np1_ba_bad_8.avi\",\"mimeType\":\"video/avi\",\"timeSegmentStart\":\"0.0s\",\"timeSegmentEnd\":\"inf\"},\"prediction\":[]}\n", + "{\"instance\":{\"content\":\"gs://automl-video-demo-data/hmdb51/_Rad_Schlag_die_Bank__cartwheel_f_cm_np1_le_med_0.avi\",\"mimeType\":\"video/avi\",\"timeSegmentStart\":\"0.0s\",\"timeSegmentEnd\":\"inf\"},\"prediction\":[{\"id\":\"7468103977691774976\",\"displayName\":\"cartwheel\",\"type\":\"shot-classification\",\"timeSegmentStart\":\"0.066666s\",\"timeSegmentEnd\":\"0.226666s\",\"confidence\":0.5290586},{\"id\":\"7468103977691774976\",\"displayName\":\"cartwheel\",\"type\":\"one-sec-interval-classification\",\"timeSegmentStart\":\"1.346666s\",\"timeSegmentEnd\":\"1.346666s\",\"confidence\":0.5290586},{\"id\":\"7468103977691774976\",\"displayName\":\"cartwheel\",\"type\":\"segment-classification\",\"timeSegmentStart\":\"0s\",\"timeSegmentEnd\":\"2.766667s\",\"confidence\":0.52444863},{\"id\":\"7468103977691774976\",\"displayName\":\"cartwheel\",\"type\":\"shot-classification\",\"timeSegmentStart\":\"0.266666s\",\"timeSegmentEnd\":\"2.226666s\",\"confidence\":0.51983875},{\"id\":\"7468103977691774976\",\"displayName\":\"cartwheel\",\"type\":\"one-sec-interval-classification\",\"timeSegmentStart\":\"1.586666s\",\"timeSegmentEnd\":\"1.586666s\",\"confidence\":0.51983875}]}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup:migration,new" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wHGVNBJSSnkx" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_model = True\n", + "delete_pipeline = True\n", + "delete_batchjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex AI fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset:\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex AI fully qualified identifier for the model\n", + "try:\n", + " if delete_model:\n", + " clients[\"model\"].delete_model(name=model_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex AI fully qualified identifier for the training pipeline\n", + "try:\n", + " if delete_pipeline:\n", + " clients[\"pipeline\"].delete_training_pipeline(name=training_pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex AI fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob:\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r gs://$BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "collapsed_sections": [ + "gcp_authenticate", + "bucket:batch_prediction", + "import_aip", + "automl_constants:automl", + "datasets_create:migration,new", + "request:migration", + "call:migration", + "response:migration", + "datasets_import:migration,new", + "MIkOh3jlSnkf", + "gOWz5vTMSnkg", + "pDdBxlyvSnkg", + "trainingpipelines_create:migration,new", + "J06YDrKMSnkh", + "AswA1FGoSnkk", + "orNAOfwiSnkk", + "trainingpipelines_get:migration,new", + "vWsdsTkRSnkm", + "HKb79nnsSnkm", + "models_evaluations_list:migration,new", + "BmzDR40DSnko", + "BnR_19BsSnko", + "models_evaluations_get:migration,new", + "1Zd-nTWkSnkp", + "XsNAiiycSnkp", + "make_batch_prediction_file:migration,new", + "make_batch_file:automl,image", + "batchpredictionjobs_create:migration,new", + "hakbUKF5Snks", + "GQtFoN4eSnkt", + "OY2yKVV_Snku", + "batchpredictionjobs_get:migration,new", + "7RdKtPhlSnkw", + "Hgt8RcLLSnkw" + ], + "name": "UJ14 unified AutoML Vision Video Classification.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/migration/UJ15 legacy AutoML Vision Video Object Tracking.ipynb b/notebooks/community/migration/UJ15 legacy AutoML Vision Video Object Tracking.ipynb new file mode 100644 index 000000000..b50933bc5 --- /dev/null +++ b/notebooks/community/migration/UJ15 legacy AutoML Vision Video Object Tracking.ipynb @@ -0,0 +1,1625 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License.\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title:migration,new" + }, + "source": [ + "# AutoML SDK: AutoML video object tracking model\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest (preview) version of AutoML SDK.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "vQWDe_6HJZJW", + "outputId": "687efd9a-e344-470e-c1fb-18bece046cf2" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-automl --user\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the Google *cloud-storage* library as well.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "5nZRXjD0JZJY", + "outputId": "4cf3eea8-952d-4222-cea1-bc631bdb8cb7" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-storage\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the Kernel\n", + "\n", + "Once you've installed the AutoML SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "3FaDxry1JZJZ" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "\n", + "if not os.getenv(\"AUTORUN\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU run-time\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your GCP project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a GCP project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the AutoML APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [Google Cloud SDK](https://cloud.google.com/sdk) is already installed in AutoML Notebooks.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" #@param {type:\"string\"}\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id", + "outputId": "1ac7bb1b-ed49-4874-f6d1-5db45456019d" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id", + "outputId": "0acbf680-cec3-44c5-8d51-1a05a81258ca" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for AutoML. We recommend when possible, to choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You cannot use a Multi-Regional Storage bucket for training with AutoML. Not all regions provide support for all AutoML services. For the latest support per region, see [Region support for AutoML services]()\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "jVMUAWrJJZJe" + }, + "outputs": [], + "source": [ + "REGION = 'us-central1' #@param {type: \"string\"}\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "LJuwNh9wJZJe" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your GCP account\n", + "\n", + "**If you are using AutoML Notebooks**, your environment is already\n", + "authenticated. Skip this step.\n", + "\n", + "*Note: If you are on an AutoML notebook and run the cell, the cell knows to skip executing the authentication steps.*\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "fEY1QcN_JZJf" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your Google Cloud account. This provides access\n", + "# to your Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on AutoML, then don't execute this code\n", + "if not os.path.exists('/opt/deeplearning/metadata/env_version'):\n", + " if 'google.colab' in sys.modules:\n", + " from google.colab import auth as google_auth\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this tutorial in a notebook locally, replace the string\n", + " # below with the path to your service account key and run this cell to\n", + " # authenticate your Google Cloud account.\n", + " else:\n", + " %env GOOGLE_APPLICATION_CREDENTIALS your_path_to_credentials.json\n", + "\n", + " # Log in to your account on Google Cloud\n", + " ! gcloud auth login\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"[your-bucket-name]\" #@param {type:\"string\"}\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n", + " BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Dc_4SqrZJZJg", + "outputId": "cfa60d02-7265-421a-b30e-dbcbe987dac8" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION gs://$BUCKET_NAME\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gkxndVXCJZJh" + }, + "outputs": [], + "source": [ + "! gsutil ls -al gs://$BUCKET_NAME\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip" + }, + "source": [ + "#### Import AutoML SDK\n", + "\n", + "Import the AutoML SDK into our Python environment.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "x3EpSgrNJZJi" + }, + "outputs": [], + "source": [ + "import json\n", + "import os\n", + "import sys\n", + "import time\n", + "\n", + "\n", + "from google.cloud import automl_v1beta1 as automl\n", + "\n", + "\n", + "from google.protobuf.json_format import MessageToJson\n", + "from google.protobuf.json_format import ParseDict\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Setup up the following constants for AutoML:\n", + "\n", + "- `PARENT`: The AutoML location root path for dataset, model and endpoint resources.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gIbQItxjJZJj" + }, + "outputs": [], + "source": [ + "# AutoML location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients" + }, + "source": [ + "## Clients\n", + "\n", + "The AutoML SDK works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the server (AutoML).\n", + "\n", + "You will use several clients in this tutorial, so set them all up upfront.\n", + "\n", + "- Dataset Service for managed datasets.\n", + "- Model Service for managed models.\n", + "- Pipeline Service for training.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving. *Note*: Prediction has a different service endpoint.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "eE3k7aqvJZJk", + "outputId": "3b280f73-9cff-4247-f8e2-8e413dfa4491" + }, + "outputs": [], + "source": [ + "def automl_client():\n", + " return automl.AutoMlClient()\n", + "\n", + "def prediction_client():\n", + " return automl.PredictionServiceClient()\n", + "\n", + "def operations_client():\n", + " return automl.AutoMlClient()._transport.operations_client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"automl\"] = automl_client()\n", + "clients[\"prediction\"] = prediction_client()\n", + "clients[\"operations\"] = operations_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:flowers,csv,icn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = 'gs://automl-video-demo-data/traffic_videos/traffic_videos.csv'\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "1OCKDsyLJZJl", + "outputId": "cb4b0db7-251a-4928-b944-ee2f109f6855" + }, + "outputs": [], + "source": [ + "! gsutil cat $IMPORT_FILE | head -n 10\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "trainingpipelines_create:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "UNASSIGNED,gs://automl-video-demo-data/traffic_videos/traffic_videos_labels.csv\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_a_dataset:migration" + }, + "source": [ + "## Create a dataset\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "mZwR43s_JZJm" + }, + "source": [ + "### [projects.locations.datasets.create](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.datasets/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "request:migration" + }, + "source": [ + "#### Request\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "datasets_create:migration,new,request", + "outputId": "e3f9d4c6-5f20-4201-a1e6-2ea6b5c3b2f1" + }, + "outputs": [], + "source": [ + "dataset = {\n", + " \"display_name\": \"traffic_\" + TIMESTAMP,\n", + " \"video_object_tracking_dataset_metadata\": {}\n", + "}\n", + "\n", + "print(MessageToJson(\n", + " automl.CreateDatasetRequest(\n", + " parent=PARENT,\n", + " dataset=dataset\n", + " ).__dict__[\"_pb\"])\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ZZFAAFAkJZJn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"dataset\": {\n", + " \"displayName\": \"traffic_20210310004803\",\n", + " \"videoObjectTrackingDatasetMetadata\": {}\n", + " }\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "call:migration" + }, + "source": [ + "#### Call\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "datasets_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].create_dataset(\n", + " parent=PARENT,\n", + " dataset=dataset\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "response:migration" + }, + "source": [ + "#### Response\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,response", + "outputId": "6b50557f-3012-4076-e5f4-7d40bb5333e2" + }, + "outputs": [], + "source": [ + "result = request\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_create:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/datasets/VOT4951119001817186304\",\n", + " \"displayName\": \"traffic_20210310004803\",\n", + " \"createTime\": \"2021-03-10T00:48:09.292248Z\",\n", + " \"etag\": \"AB3BwFp3u4a2oy-k3EgK6ci8zwrTqrd91_DmoaY8TYsxnb-N-aXwFefqCIm1z0YTM290\",\n", + " \"videoObjectTrackingDatasetMetadata\": {}\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:migration,new,response", + "outputId": "ddb296ce-8c86-4793-a7ba-b5409a7d92d1" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split('/')[-1]\n", + "\n", + "print(dataset_id)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "OOmLNEw3JZJq" + }, + "source": [ + "### [projects.locations.datasets.importData](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.datasets/importData)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "T5yWOG67JZJr" + }, + "source": [ + "#### Request\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "datasets_import:migration,new,request", + "outputId": "5cac21ee-6f18-4f44-c360-d174832d32ff" + }, + "outputs": [], + "source": [ + "input_config = {\n", + " \"gcs_source\": {\n", + " \"input_uris\": [IMPORT_FILE]\n", + " }\n", + "}\n", + "\n", + "print(MessageToJson(\n", + " automl.ImportDataRequest(\n", + " name=dataset_short_id,\n", + " input_config=input_config\n", + " ).__dict__[\"_pb\"])\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Dzg9JangJZJs" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"VOT4951119001817186304\",\n", + " \"inputConfig\": {\n", + " \"gcsSource\": {\n", + " \"inputUris\": [\n", + " \"gs://automl-video-demo-data/traffic_videos/traffic_videos.csv\"\n", + " ]\n", + " }\n", + " }\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "1shZi6Y4JZJs" + }, + "source": [ + "#### Call\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "datasets_import:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].import_data(\n", + " name=dataset_id,\n", + " input_config=input_config\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "o3_x634HJZJt" + }, + "source": [ + "#### Response\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Dc3YtK0jJZJt", + "outputId": "235c07d2-b693-48aa-a8af-ffc087541ab3" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_a_model:migration" + }, + "source": [ + "## Train a model\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_models_create:migration,old" + }, + "source": [ + "### [projects.locations.models.create](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "mQD_PyZjJZJu" + }, + "source": [ + "#### Request\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_create:migration,new,request,icn", + "outputId": "e900bb95-0e65-4576-efc6-e1ef8bbf7a28" + }, + "outputs": [], + "source": [ + "model = {\n", + " \"display_name\": \"traffic_\" + TIMESTAMP,\n", + " \"dataset_id\": dataset_short_id,\n", + " \"video_object_tracking_model_metadata\": {}\n", + "}\n", + "\n", + "print(MessageToJson(\n", + " automl.CreateModelRequest(\n", + " parent=PARENT,\n", + " model=model\n", + " ).__dict__[\"_pb\"])\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "8atEKPpdJZJv" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"model\": {\n", + " \"displayName\": \"traffic_20210310004803\",\n", + " \"datasetId\": \"VOT4951119001817186304\",\n", + " \"videoObjectTrackingModelMetadata\": {}\n", + " }\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ZfHEsMO5JZJv" + }, + "source": [ + "#### Call\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].create_model(\n", + " parent=PARENT,\n", + " model=model\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "jYKM6gfvJZJw" + }, + "source": [ + "#### Response\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,request", + "outputId": "80d54292-09b6-4123-d6eb-0450ce8eb816" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "jdU2ko30JZJw" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/VOT6634816000837550080\"\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "training_pipeline_id:migration,new,response", + "outputId": "30468e96-7513-4bf3-decc-3b0ea10534a5" + }, + "outputs": [], + "source": [ + "# The full unique ID for the training pipeline\n", + "model_id = result.name\n", + "# The short numeric ID for the training pipeline\n", + "model_short_id = model_id.split('/')[-1]\n", + "\n", + "print(model_short_id)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:migration" + }, + "source": [ + "## Evaluate the model\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "yfufMwAEJjkX" + }, + "source": [ + "### [projects.locations.models.modelEvaluations.list](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models.modelEvaluations/list)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "_Zv5eHjwJZJx" + }, + "source": [ + "#### Call\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "models_evaluations_list:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].list_model_evaluations(\n", + " parent=model_id, \n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "QtnV8mnJJZJx" + }, + "source": [ + "#### Response\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "models_evaluations_list:migration,new,response", + "outputId": "8a5080de-5c4f-41f8-8930-71dd7afd8405", + "scrolled": true + }, + "outputs": [], + "source": [ + "for evaluation in request:\n", + " print(MessageToJson(evaluation.__dict__[\"_pb\"]))\n", + "\n", + "# The last evaluation slice\n", + "last_evaluation_slice = evaluation.name\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "72tAHlBJJZJy" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/VOT6634816000837550080/modelEvaluations/3861662240237183015\",\n", + " \"annotationSpecId\": \"7080515133784457216\",\n", + " \"createTime\": \"2021-03-10T01:57:44.615737Z\",\n", + " \"evaluatedExampleCount\": 6,\n", + " \"videoObjectTrackingEvaluationMetrics\": {\n", + " \"boundingBoxMetricsEntries\": [\n", + " {\n", + " \"iouThreshold\": 0.5,\n", + " \"meanAveragePrecision\": 0.30026233,\n", + " \"confidenceMetricsEntries\": [\n", + " {\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.37222221,\n", + " \"f1Score\": 0.5425101\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.028951555,\n", + " \"recall\": 0.13432837,\n", + " \"precision\": 0.07377049,\n", + " \"f1Score\": 0.0952381\n", + " },\n", + " \n", + " # REMOVED FOR BREVITY\n", + " \n", + " {\n", + " \"confidenceThreshold\": 1.0,\n", + " \"precision\": 1.0\n", + " }\n", + " ]\n", + " }\n", + " ],\n", + " \"boundingBoxMeanAveragePrecision\": 0.30026233\n", + " },\n", + " \"displayName\": \"pickup_suv_van\"\n", + "}\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/VOT6634816000837550080/modelEvaluations/4053501621797068779\",\n", + " \"annotationSpecId\": \"5927593629177610240\",\n", + " \"createTime\": \"2021-03-10T01:57:44.615737Z\",\n", + " \"evaluatedExampleCount\": 5,\n", + " \"videoObjectTrackingEvaluationMetrics\": {\n", + " \"boundingBoxMetricsEntries\": [\n", + " {\n", + " \"iouThreshold\": 0.5,\n", + " \"meanAveragePrecision\": 0.42889464,\n", + " \"confidenceMetricsEntries\": [\n", + " {\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.25490198,\n", + " \"f1Score\": 0.40625\n", + " },\n", + " \n", + " # REMOVED FOR BREVITY\n", + " ],\n", + " \"boundingBoxMeanAveragePrecision\": 0.34359422\n", + " },\n", + " \"displayName\": \"sedan\"\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "i6T0bzuNJjkY" + }, + "source": [ + "### [projects.locations.models.modelEvaluations.get](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models.modelEvaluations/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Nusdm5_MJZJz" + }, + "source": [ + "#### Call\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "models_evaluations_get:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].get_model_evaluation(\n", + " name=last_evaluation_slice\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "GB8oloXGJZJz" + }, + "source": [ + "#### Response\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "9n1UTT6oJZJ0", + "outputId": "898d5c96-225c-4811-8504-450662cd11fb", + "scrolled": true + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "vqQCboLGJZJ0" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/VOT6634816000837550080/modelEvaluations/6603022638609369541\",\n", + " \"annotationSpecId\": \"1315907610750222336\",\n", + " \"createTime\": \"2021-03-10T01:57:44.615737Z\",\n", + " \"evaluatedExampleCount\": 6,\n", + " \"videoObjectTrackingEvaluationMetrics\": {\n", + " \"boundingBoxMetricsEntries\": [\n", + " {\n", + " \"iouThreshold\": 0.5,\n", + " \"meanAveragePrecision\": 0.34359422,\n", + " \"confidenceMetricsEntries\": [\n", + " {\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.41428572,\n", + " \"f1Score\": 0.5858586\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.03328514,\n", + " \"recall\": 0.1724138,\n", + " \"precision\": 0.10869565,\n", + " \"f1Score\": 0.13333334\n", + " },\n", + " \n", + " # REMOVED FOR BREVITY\n", + " ]\n", + " }\n", + " ],\n", + " \"boundingBoxMeanAveragePrecision\": 0.34359422\n", + " },\n", + " \"displayName\": \"sedan\"\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_predictions:migration" + }, + "source": [ + "## Make batch predictions\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_prediction_file:migration,new" + }, + "source": [ + "### Prepare batch prediction data\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_items:automl,icn,csv", + "outputId": "95539c27-5e06-4d5b-c7ec-c9d199c2c713" + }, + "outputs": [], + "source": [ + "TRAIN_FILES = \"gs://automl-video-demo-data/traffic_videos/traffic_videos_labels.csv\"\n", + "\n", + "test_items = ! gsutil cat $TRAIN_FILES | head -n2\n", + "\n", + "cols = str(test_items[0]).split(',')\n", + "test_item_1 = str(cols[0])\n", + "test_label_1 = str(cols[1])\n", + "test_start_time_1 = str(0)\n", + "test_end_time_1 = \"inf\"\n", + "\n", + "print(test_item_1, test_label_1)\n", + "\n", + "cols = str(test_items[1]).split(',')\n", + "test_item_2 = str(cols[0])\n", + "test_label_2 = str(cols[1])\n", + "test_start_time_2 = str(0)\n", + "test_end_time_2 = \"inf\"\n", + "\n", + "print(test_item_2, test_label_2)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "mrOPGyQ-JZJ2" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://automl-video-demo-data/traffic_videos/highway_005.mp4 sedan\n", + "gs://automl-video-demo-data/traffic_videos/highway_005.mp4 pickup_suv_van\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:automl,image" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "To request a batch of predictions from AutoML Video, create a CSV file that lists the Cloud Storage paths to the videos that you want to annotate. You can also specify a start and end time to tell AutoML Video to only annotate a segment (segment-level) of the video. The start time must be zero or greater and must be before the end time. The end time must be greater than the start time and less than or equal to the duration of the video. You can also use inf to indicate the end of a video.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "_EdGkXQPJZJ2", + "outputId": "e9c08c85-e220-47b2-9081-54034471da6f" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "import json\n", + "\n", + "gcs_input_uri = \"gs://\" + BUCKET_NAME + '/test.csv'\n", + "with tf.io.gfile.GFile(gcs_input_uri, 'w') as f:\n", + " data = f\"{test_item_1}, {test_start_time_1}, {test_end_time_1}\"\n", + " f.write(data + '\\n')\n", + " data = f\"{test_item_2}, {test_start_time_2}, {test_end_time_2}\"\n", + " f.write(data + '\\n')\n", + " \n", + "print(gcs_input_uri)\n", + "!gsutil cat $gcs_input_uri\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "PEIM210uJZJ3" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210310004803/test.csv\n", + "gs://automl-video-demo-data/traffic_videos/highway_005.mp4, 0, inf\n", + "gs://automl-video-demo-data/traffic_videos/highway_005.mp4, 0, inf\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_models_batchpredict:migration,old" + }, + "source": [ + "### [projects.locations.models.batchPredict](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models/batchPredict)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "zKU3o21IJZJ4" + }, + "source": [ + "#### Request\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_create:migration,new,request,icn", + "outputId": "74db149d-12c0-4a96-b99e-a22d226d06ac" + }, + "outputs": [], + "source": [ + "input_config = {\n", + " \"gcs_source\": {\n", + " \"input_uris\": [gcs_input_uri]\n", + " }\n", + "}\n", + " \n", + "output_config = {\n", + " \"gcs_destination\": {\n", + " \"output_uri_prefix\": \"gs://\" + f\"{BUCKET_NAME}/batch_output/\"\n", + " }\n", + "} \n", + "\n", + "batch_prediction = automl.BatchPredictRequest(\n", + " name=model_id,\n", + " input_config=input_config,\n", + " output_config=output_config,\n", + ")\n", + "\n", + "print(MessageToJson(batch_prediction.__dict__[\"_pb\"]))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "1Y0TXjVNJZJ4" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/VOT6634816000837550080\",\n", + " \"inputConfig\": {\n", + " \"gcsSource\": {\n", + " \"inputUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210310004803/test.csv\"\n", + " ]\n", + " }\n", + " },\n", + " \"outputConfig\": {\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210310004803/batch_output/\"\n", + " }\n", + " }\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ScGH6fbbJZJ5" + }, + "source": [ + "#### Call\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"prediction\"].batch_predict(\n", + " request=batch_prediction\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Mh9Jwl8OJZJ5" + }, + "source": [ + "#### Response\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ep-WdlhxJZJ6", + "outputId": "7f3a4e85-0700-4648-a380-4be7a0da2c3d" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_create:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "\n", + "```\n", + "{}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "GW6GcRaaJZJ6", + "outputId": "5ea11a6f-226c-4ba7-e274-e9f09ecdce48" + }, + "outputs": [], + "source": [ + "destination_uri = batch_prediction.output_config.gcs_destination.output_uri_prefix[:-1]\n", + "\n", + "! gsutil ls $destination_uri/prediction-**\n", + "! gsutil cat $destination_uri/prediction-**" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gy3AJcK5JZJ7" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210310004803/batch_output/prediction-traffic_20210310004803-2021-03-10T01:57:52.257240Z/highway_005_1.json\n", + "gs://migration-ucaip-trainingaip-20210310004803/batch_output/prediction-traffic_20210310004803-2021-03-10T01:57:52.257240Z/video_object_tracking.csv\n", + "{\n", + " \"object_annotations\": [ {\n", + " \"annotation_spec\": {\n", + " \"display_name\": \"sedan\",\n", + " \"description\": \"sedan\"\n", + " },\n", + " \"confidence\": 0.52724433,\n", + " \"frames\": [ {\n", + " \"normalized_bounding_box\": {\n", + " \"x_min\": 0.27629745,\n", + " \"y_min\": 0.59244406,\n", + " \"x_max\": 0.53941643,\n", + " \"y_max\": 0.77127469\n", + " },\n", + " \"time_offset\": {\n", + " \n", + " }\n", + " }, {\n", + " \"normalized_bounding_box\": {\n", + " \"x_min\": 0.135607,\n", + " \"y_min\": 0.58437037,\n", + " \"x_max\": 0.42441425,\n", + " \"y_max\": 0.77325606\n", + " },\n", + " \"time_offset\": {\n", + " \"nanos\": 100000000\n", + " }\n", + " }, \n", + " \n", + " # REMOVED FOR BREVITY\n", + " \n", + " }\n", + " } ]\n", + "}\n", + "gs://automl-video-demo-data/traffic_videos/highway_005.mp4,0,315576000000,gs://migration-ucaip-trainingaip-20210310004803/batch_output/prediction-traffic_20210310004803-2021-03-10T01:57:52.257240Z/highway_005_1.json,OK\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup:migration,new" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "panMCMF8JZJ8", + "outputId": "737f9f35-650f-4329-8b0e-8a9dfe026a1a" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_model = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the AutoML fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset:\n", + " clients['automl'].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the AutoML fully qualified identifier for the model\n", + "try:\n", + " if delete_model:\n", + " clients['automl'].delete_model(name=model_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and 'BUCKET_NAME' in globals():\n", + " ! gsutil rm -r gs://$BUCKET_NAME\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Z9fbty3uJZJ8" + }, + "outputs": [], + "source": [] + } + ], + "metadata": { + "colab": { + "collapsed_sections": [ + "gcp_authenticate", + "bucket:batch_prediction", + "setup_vars", + "import_aip", + "aip_constants", + "mZwR43s_JZJm", + "request:migration", + "call:migration", + "response:migration", + "OOmLNEw3JZJq", + "T5yWOG67JZJr", + "1shZi6Y4JZJs", + "o3_x634HJZJt", + "text_models_create:migration,old", + "mQD_PyZjJZJu", + "ZfHEsMO5JZJv", + "jYKM6gfvJZJw", + "yfufMwAEJjkX", + "_Zv5eHjwJZJx", + "QtnV8mnJJZJx", + "i6T0bzuNJjkY", + "Nusdm5_MJZJz", + "GB8oloXGJZJz", + "make_batch_prediction_file:migration,new", + "text_models_batchpredict:migration,old", + "zKU3o21IJZJ4", + "ScGH6fbbJZJ5", + "Mh9Jwl8OJZJ5" + ], + "name": "[UJ.15 OLD] AutoML Vision Video Object Tracking.ipynb", + "provenance": [], + "toc_visible": true + }, + "environment": { + "name": "tf2-2-3-gpu.2-3.m55", + "type": "gcloud", + "uri": "gcr.io/deeplearning-platform-release/tf2-2-3-gpu.2-3:m55" + }, + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.7.8" + } + }, + "nbformat": 4, + "nbformat_minor": 4 +} diff --git a/notebooks/community/migration/UJ15 unified AutoML Vision Video Object Tracking.ipynb b/notebooks/community/migration/UJ15 unified AutoML Vision Video Object Tracking.ipynb new file mode 100644 index 000000000..f12b231bf --- /dev/null +++ b/notebooks/community/migration/UJ15 unified AutoML Vision Video Object Tracking.ipynb @@ -0,0 +1,1902 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title:migration,new" + }, + "source": [ + "# Vertex SDK: AutoML video object tracking model\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest (preview) version of Vertex SDK.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "KDLhKMzGn5Hx" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-aiplatform --user" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the Google *cloud-storage* library as well.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "XKUq0zsKn5H3" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-storage" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the Kernel\n", + "\n", + "Once you've installed the Vertex SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "kvM_QWtcn5H5" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "if not os.getenv(\"AUTORUN\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU run-time\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your GCP project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a GCP project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebooks.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex AI. We recommend when possible, to choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You cannot use a Multi-Regional Storage bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see [Region support for Vertex AI services](https://cloud.google.com/vertex-ai/docs/general/locations)\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pTJUl6sSn5H7" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Gn2q7I4yn5H8" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your GCP account\n", + "\n", + "**If you are using Google Cloud Notebooks**, your environment is already\n", + "authenticated. Skip this step.\n", + "\n", + "*Note: If you are on an Vertex notebook and run the cell, the cell knows to skip executing the authentication steps.*\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "TFasBw7Cn5H9" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your Google Cloud account. This provides access\n", + "# to your Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Vertex, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this tutorial in a notebook locally, replace the string\n", + " # below with the path to your service account key and run this cell to\n", + " # authenticate your Google Cloud account.\n", + " else:\n", + " %env GOOGLE_APPLICATION_CREDENTIALS your_path_to_credentials.json\n", + "\n", + " # Log in to your account on Google Cloud\n", + " ! gcloud auth login" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n", + " BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "KTRHxxatn5H-" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "OBbXRqZPn5H_" + }, + "outputs": [], + "source": [ + "! gsutil ls -al gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip" + }, + "source": [ + "#### Import Vertex SDK\n", + "\n", + "Import the Vertex SDK into our Python environment.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Z7wda4l9n5IA" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex AI constants\n", + "\n", + "Setup up the following constants for Vertex AI:\n", + "\n", + "- `API_ENDPOINT`: The Vertex AI API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex AI location root path for dataset, model and endpoint resources.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "T6CQX7uwn5IA" + }, + "outputs": [], + "source": [ + "# API Endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex AI location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants:automl" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Next, setup constants unique to AutoML video object tracking datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the managed dataset service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the managed dataset service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the managed pipelines service the task (e.g., classification) to train the model for.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:automl,icn" + }, + "outputs": [], + "source": [ + "# Video Dataset type\n", + "VIDEO_SCHEMA = \"google-cloud-aiplatform/schema/dataset/metadata/video_1.0.0.yaml\"\n", + "# Video Labeling type\n", + "IMPORT_SCHEMA_VIDEO_OBJECT_TRACKING = \"gs://google-cloud-aiplatform/schema/dataset/ioformat/video_object_tracking_io_format_1.0.0.yaml\"\n", + "# Video Training task\n", + "TRAINING_VIDEO_OBJECT_TRACKING_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_video_object_tracking_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients" + }, + "source": [ + "## Clients\n", + "\n", + "The Vertex SDK works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the server (Vertex).\n", + "\n", + "You will use several clients in this tutorial, so set them all up upfront.\n", + "\n", + "- Dataset Service for managed datasets.\n", + "- Model Service for managed models.\n", + "- Pipeline Service for training.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving. *Note*: Prediction has a different service endpoint.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "H8mUvtiTn5IB" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "clients[\"job\"] = create_job_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:flowers,csv,icn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://automl-video-demo-data/traffic_videos/traffic_videos_labels.csv\"" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ptL5nC02n5IC" + }, + "outputs": [], + "source": [ + "! gsutil cat $IMPORT_FILE | head -n 10" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "trainingpipelines_create:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://automl-video-demo-data/traffic_videos/highway_005.mp4,sedan,1565750291672021,11.933333,0.509205,0.594283,,,0.728737,0.760959,,\n", + "gs://automl-video-demo-data/traffic_videos/highway_005.mp4,pickup_suv_van,1565750291672171,17.566666,0.761241,0.498466,,,0.948839,0.668524,,\n", + "gs://automl-video-demo-data/traffic_videos/highway_005.mp4,pickup_suv_van,1565750291672223,20.433333,0.000000,0.465235,,,0.142638,0.665644,,\n", + "gs://automl-video-demo-data/traffic_videos/highway_005.mp4,pickup_suv_van,1565750291672347,25.766666,0.486523,0.592331,,,0.720611,0.776687,,\n", + "gs://automl-video-demo-data/traffic_videos/highway_005.mp4,pickup_suv_van,1565750291672575,28.966666,0.578534,0.652778,,,0.828647,0.862967,,\n", + "gs://automl-video-demo-data/traffic_videos/highway_005.mp4,pickup_suv_van,1565750291672549,28.966666,0.000000,0.518571,,,0.148841,0.737677,,\n", + "gs://automl-video-demo-data/traffic_videos/highway_005.mp4,pickup_suv_van,1565750291672599,28.966666,0.106979,0.458078,,,0.377877,0.678937,,\n", + "gs://automl-video-demo-data/traffic_videos/highway_005.mp4,pickup_suv_van,1565715798494273,32.466666,0.333083,0.485473,,,0.542722,0.647774,,\n", + "gs://automl-video-demo-data/traffic_videos/highway_005.mp4,sedan,1565715798494439,36.433333,0.935638,0.564839,,,1.000000,0.672182,,\n", + "gs://automl-video-demo-data/traffic_videos/highway_005.mp4,pickup_suv_van,1565715798494381,36.433333,0.000000,0.455703,,,0.164878,0.660083,,\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_a_dataset:migration" + }, + "source": [ + "## Create a dataset\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_create:migration,new" + }, + "source": [ + "### [projects.locations.datasets.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.datasets/create)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "request:migration" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "datasets_create:migration,new,request" + }, + "outputs": [], + "source": [ + "DATA_SCHEMA = VIDEO_SCHEMA\n", + "\n", + "dataset = {\n", + " \"display_name\": \"traffic_\" + TIMESTAMP,\n", + " \"metadata_schema_uri\": \"gs://\" + DATA_SCHEMA,\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateDatasetRequest(parent=PARENT, dataset=dataset).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"dataset\": {\n", + " \"displayName\": \"traffic_20210310013516\",\n", + " \"metadataSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/metadata/video_1.0.0.yaml\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "call:migration" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "datasets_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "response:migration" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,response" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "OC-Yc89In5IH" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/datasets/7534187925055995904\",\n", + " \"displayName\": \"traffic_20210310013516\",\n", + " \"metadataSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/metadata/video_1.0.0.yaml\",\n", + " \"labels\": {\n", + " \"aiplatform.googleapis.com/dataset_metadata_schema\": \"VIDEO\"\n", + " },\n", + " \"metadata\": {\n", + " \"dataItemSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/dataitem/video_1.0.0.yaml\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new" + }, + "source": [ + "### [projects.locations.datasets.import](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.datasets/import)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "qYXlUMfWn5II" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "outputs": [], + "source": [ + "LABEL_SCHEMA = IMPORT_SCHEMA_VIDEO_OBJECT_TRACKING\n", + "\n", + "import_config = {\n", + " \"gcs_source\": {\"uris\": [IMPORT_FILE]},\n", + " \"import_schema_uri\": LABEL_SCHEMA,\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.ImportDataRequest(\n", + " name=dataset_short_id, import_configs=[import_config]\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "AVJY95yXn5II" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"7534187925055995904\",\n", + " \"importConfigs\": [\n", + " {\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://automl-video-demo-data/traffic_videos/traffic_videos_labels.csv\"\n", + " ]\n", + " },\n", + " \"importSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/ioformat/video_object_tracking_io_format_1.0.0.yaml\"\n", + " }\n", + " ]\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "2qSg5OOWn5II" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "datasets_import:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"dataset\"].import_data(\n", + " name=dataset_id, import_configs=[import_config]\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "y6bxHeSWn5IJ" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ojm6e7dTn5IJ" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sMdZTOKsn5IK" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_a_model:migration" + }, + "source": [ + "## Train a model\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "trainingpipelines_create:migration,new" + }, + "source": [ + "### [projects.locations.trainingPipelines.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.trainingPipelines/create)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ZJvJeQ5mn5IK" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_create:migration,new,request,icn" + }, + "outputs": [], + "source": [ + "TRAINING_SCHEMA = TRAINING_VIDEO_OBJECT_TRACKING_SCHEMA\n", + "\n", + "task = Value(struct_value=Struct(fields={\"model_type\": Value(string_value=\"CLOUD\")}))\n", + "\n", + "training_pipeline = {\n", + " \"display_name\": \"traffic_\" + TIMESTAMP,\n", + " \"training_task_definition\": TRAINING_SCHEMA,\n", + " \"training_task_inputs\": task,\n", + " \"input_data_config\": {\n", + " \"dataset_id\": dataset_short_id,\n", + " \"fraction_split\": {\"training_fraction\": 0.8, \"test_fraction\": 0.2},\n", + " },\n", + " \"model_to_upload\": {\"display_name\": \"traffic_\" + TIMESTAMP},\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateTrainingPipelineRequest(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "SMWsytANn5IK" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"trainingPipeline\": {\n", + " \"displayName\": \"traffic_20210310013516\",\n", + " \"inputDataConfig\": {\n", + " \"datasetId\": \"7534187925055995904\",\n", + " \"fractionSplit\": {\n", + " \"trainingFraction\": 0.8,\n", + " \"testFraction\": 0.2\n", + " }\n", + " },\n", + " \"trainingTaskDefinition\": \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_video_object_tracking_1.0.0.yaml\",\n", + " \"trainingTaskInputs\": {\n", + " \"model_type\": \"CLOUD\"\n", + " },\n", + " \"modelToUpload\": {\n", + " \"displayName\": \"traffic_20210310013516\"\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "SdESf_Xjn5IL" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "jk2KsljSn5IL" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,request" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gUIK_1mcn5IM" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/trainingPipelines/4612961451915608064\",\n", + " \"displayName\": \"traffic_20210310013516\",\n", + " \"inputDataConfig\": {\n", + " \"datasetId\": \"7534187925055995904\",\n", + " \"fractionSplit\": {\n", + " \"trainingFraction\": 0.8,\n", + " \"testFraction\": 0.2\n", + " }\n", + " },\n", + " \"trainingTaskDefinition\": \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_video_object_tracking_1.0.0.yaml\",\n", + " \"trainingTaskInputs\": {\n", + " \"modelType\": \"CLOUD\"\n", + " },\n", + " \"modelToUpload\": {\n", + " \"displayName\": \"traffic_20210310013516\"\n", + " },\n", + " \"state\": \"PIPELINE_STATE_PENDING\",\n", + " \"createTime\": \"2021-03-10T13:09:36.473816Z\",\n", + " \"updateTime\": \"2021-03-10T13:09:36.473816Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "training_pipeline_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the training pipeline\n", + "training_pipeline_id = request.name\n", + "# The short numeric ID for the training pipeline\n", + "training_pipeline_short_id = training_pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(training_pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "trainingpipelines_get:migration,new" + }, + "source": [ + "### [projects.locations.trainingPipelines.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.trainingPipelines/get)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "egJoW5sUn5IM" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_get:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"pipeline\"].get_training_pipeline(name=training_pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "INONfT8Ln5IN" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "xYKvFpAVn5IN" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "eeT7dtSvn5IN" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/trainingPipelines/4612961451915608064\",\n", + " \"displayName\": \"traffic_20210310013516\",\n", + " \"inputDataConfig\": {\n", + " \"datasetId\": \"7534187925055995904\",\n", + " \"fractionSplit\": {\n", + " \"trainingFraction\": 0.8,\n", + " \"testFraction\": 0.2\n", + " }\n", + " },\n", + " \"trainingTaskDefinition\": \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_video_object_tracking_1.0.0.yaml\",\n", + " \"trainingTaskInputs\": {\n", + " \"modelType\": \"CLOUD\"\n", + " },\n", + " \"modelToUpload\": {\n", + " \"displayName\": \"traffic_20210310013516\"\n", + " },\n", + " \"state\": \"PIPELINE_STATE_PENDING\",\n", + " \"createTime\": \"2021-03-10T13:09:36.473816Z\",\n", + " \"updateTime\": \"2021-03-10T13:09:36.473816Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=training_pipeline_id)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_name = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " break\n", + " else:\n", + " model_id = response.model_to_upload.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:migration" + }, + "source": [ + "## Evaluate the model\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "models_evaluations_list:migration,new" + }, + "source": [ + "### [projects.locations.models.evaluations.list](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.models.evaluations/list)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "xTM1xhTzn5IO" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "models_evaluations_list:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"model\"].list_model_evaluations(parent=model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "qRTiZUwYn5IO" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "AxEY7WFHj2YC" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "model_evaluations = [json.loads(MessageToJson(me.__dict__[\"_pb\"])) for me in request]\n", + "# The evaluation slice\n", + "evaluation_slice = request.model_evaluations[0].name\n", + "\n", + "print(json.dumps(model_evaluations, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "rBPb8NvWn5IP" + }, + "source": [ + "*Example output*:\n", + "```\n", + "[\n", + " {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/6125898247828406272/evaluations/305090287452028928\",\n", + " \"metricsSchemaUri\": \"gs://google-cloud-aiplatform/schema/modelevaluation/video_object_tracking_metrics_1.0.0.yaml\",\n", + " \"metrics\": {\n", + " \"boundingBoxMetrics\": [\n", + " {\n", + " \"meanAveragePrecision\": 0.34263912,\n", + " \"iouThreshold\": 0.5,\n", + " \"confidenceMetrics\": [\n", + " {\n", + " \"precision\": 0.36842105,\n", + " \"recall\": 1.0,\n", + " \"f1Score\": 0.53846157\n", + " },\n", + " {\n", + " \"precision\": 0.088,\n", + " \"confidenceThreshold\": 0.032954127,\n", + " \"recall\": 0.16541353,\n", + " \"f1Score\": 0.11488251\n", + " },\n", + " {\n", + " \"precision\": 0.08835341,\n", + " \"confidenceThreshold\": 0.035069585,\n", + " \"recall\": 0.16541353,\n", + " \"f1Score\": 0.11518325\n", + " },\n", + " {\n", + " \"precision\": 0.088709675,\n", + " \"recall\": 0.16541353,\n", + " \"confidenceThreshold\": 0.036181003,\n", + " \"f1Score\": 0.115485564\n", + " },\n", + " {\n", + " \"recall\": 0.16541353,\n", + " \"f1Score\": 0.11578947,\n", + " \"confidenceThreshold\": 0.037186295,\n", + " \"precision\": 0.08906882\n", + " },\n", + " {\n", + " \"recall\": 0.16541353,\n", + " \"precision\": 0.08943089,\n", + " \"confidenceThreshold\": 0.038205147,\n", + " \"f1Score\": 0.116094984\n", + " },\n", + " \n", + " # REMOVED FOR BREVITY\n", + " {\n", + " \"recall\": 0.007518797,\n", + " \"precision\": 1.0,\n", + " \"confidenceThreshold\": 0.66486305,\n", + " \"f1Score\": 0.014925373\n", + " },\n", + " {\n", + " \"precision\": 1.0,\n", + " \"confidenceThreshold\": 1.0\n", + " }\n", + " ]\n", + " }\n", + " ],\n", + " \"boundingBoxMeanAveragePrecision\": 0.34263912\n", + " },\n", + " \"createTime\": \"2021-03-10T14:18:31.880535Z\",\n", + " \"sliceDimensions\": [\n", + " \"annotationSpec\"\n", + " ]\n", + " }\n", + "]\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "models_evaluations_get:migration,new" + }, + "source": [ + "### [projects.locations.models.evaluations.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.models.evaluations/get)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Othev0Hnn5IP" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "models_evaluations_get:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"model\"].get_model_evaluation(name=evaluation_slice)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "LUV4WStsn5IQ" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pktv6Vcxn5IQ" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "v9HtXNL4n5IQ" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/6125898247828406272/evaluations/305090287452028928\",\n", + " \"metricsSchemaUri\": \"gs://google-cloud-aiplatform/schema/modelevaluation/video_object_tracking_metrics_1.0.0.yaml\",\n", + " \"metrics\": {\n", + " \"boundingBoxMetrics\": [\n", + " {\n", + " \"confidenceMetrics\": [\n", + " {\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.36842105,\n", + " \"f1Score\": 0.53846157\n", + " },\n", + " {\n", + " \"recall\": 0.16541353,\n", + " \"precision\": 0.088,\n", + " \"f1Score\": 0.11488251,\n", + " \"confidenceThreshold\": 0.032954127\n", + " },\n", + " \n", + " # REMOVED FOR BREVITY\n", + " \n", + " {\n", + " \"confidenceThreshold\": 1.0,\n", + " \"precision\": 1.0\n", + " }\n", + " ],\n", + " \"meanAveragePrecision\": 0.34263912,\n", + " \"iouThreshold\": 0.5\n", + " }\n", + " ],\n", + " \"boundingBoxMeanAveragePrecision\": 0.34263912\n", + " },\n", + " \"createTime\": \"2021-03-10T14:18:31.880535Z\",\n", + " \"sliceDimensions\": [\n", + " \"annotationSpec\"\n", + " ]\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_predictions:migration" + }, + "source": [ + "## Make batch predictions\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_prediction_file:migration,new" + }, + "source": [ + "### Prepare batch prediction data\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_items:automl,icn,csv" + }, + "outputs": [], + "source": [ + "test_items = ! gsutil cat $IMPORT_FILE | head -n25\n", + "\n", + "cols_1 = test_items[0].split(\",\")\n", + "cols_2 = test_items[-1].split(\",\")\n", + "\n", + "if len(cols_1) > 12:\n", + " test_item_1 = str(cols_1[1])\n", + " test_item_2 = str(cols_2[1])\n", + " test_label_1 = str(cols_1[5:])\n", + " test_label_2 = str(cols_2[5:])\n", + "else:\n", + " test_item_1 = str(cols_1[0])\n", + " test_item_2 = str(cols_2[0])\n", + " test_label_1 = str(cols_1[4:])\n", + " test_label_2 = str(cols_2[4:])\n", + "\n", + "\n", + "print(test_item_1, test_label_1)\n", + "print(test_item_2, test_label_2)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "6fRc0HuWn5IS" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://automl-video-demo-data/traffic_videos/highway_005.mp4 ['0.509205', '0.594283', '', '', '0.728737', '0.760959', '', '']\n", + "gs://automl-video-demo-data/traffic_videos/highway_006.mp4 ['0.621857', '0.561570', '', '', '0.825726', '0.699151', '', '']\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:automl,image" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "Let's now make a batch input file, which you store in your local Cloud Storage bucket. The batch input file can be either CSV or JSONL. You will use JSONL in this tutorial. For JSONL file, you make one dictionary entry per line for each video. The dictionary contains the key/value pairs:\n", + "\n", + "- `content`: The Cloud Storage path to the video.\n", + "- `mimeType`: The content type. In our example, it is an `avi` file.\n", + "- `timeSegmentStart`: The start timestamp in the video to do prediction on. *Note*, the timestamp must be specified as a string and followed by s (second), m (minute) or h (hour).\n", + "- `timeSegmentEnd`: The end timestamp in the video to do prediction on.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "0rWoXOq9n5IS" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "import tensorflow as tf\n", + "\n", + "gcs_input_uri = \"gs://\" + BUCKET_NAME + \"/test.jsonl\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " data = {\n", + " \"content\": test_item_1,\n", + " \"mimeType\": \"video/avi\",\n", + " \"timeSegmentStart\": \"0.0s\",\n", + " \"timeSegmentEnd\": \"inf\",\n", + " }\n", + " f.write(json.dumps(data) + \"\\n\")\n", + " data = {\n", + " \"content\": test_item_2,\n", + " \"mimeType\": \"video/avi\",\n", + " \"timeSegmentStart\": \"0.0s\",\n", + " \"timeSegmentEnd\": \"inf\",\n", + " }\n", + " f.write(json.dumps(data) + \"\\n\")\n", + "\n", + "print(gcs_input_uri)\n", + "!gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "trainingpipelines_get:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210310013516/test.jsonl\n", + "{\"content\": \"gs://automl-video-demo-data/traffic_videos/highway_005.mp4\", \"mimeType\": \"video/avi\", \"timeSegmentStart\": \"0.0s\", \"timeSegmentEnd\": \"inf\"}\n", + "{\"content\": \"gs://automl-video-demo-data/traffic_videos/highway_006.mp4\", \"mimeType\": \"video/avi\", \"timeSegmentStart\": \"0.0s\", \"timeSegmentEnd\": \"inf\"}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_create:migration,new" + }, + "source": [ + "### [projects.locations.batchPredictionJobs.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.batchPredictionJobs/create)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "S5Odu9ogn5IS" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_create:migration,new,request,icn" + }, + "outputs": [], + "source": [ + "batch_prediction_job = {\n", + " \"display_name\": \"traffic_\" + TIMESTAMP,\n", + " # Format: 'projects/{project}/locations/{location}/models/{model_id}'\n", + " \"model\": model_id,\n", + " \"model_parameters\": json_format.ParseDict(\n", + " {\"confidenceThreshold\": 0.5, \"maxPredictions\": 2}, Value()\n", + " ),\n", + " \"input_config\": {\n", + " \"instances_format\": \"jsonl\",\n", + " \"gcs_source\": {\"uris\": [gcs_input_uri]},\n", + " },\n", + " \"output_config\": {\n", + " \"predictions_format\": \"jsonl\",\n", + " \"gcs_destination\": {\n", + " \"output_uri_prefix\": \"gs://\" + f\"{BUCKET_NAME}/batch_output/\"\n", + " },\n", + " },\n", + " \"dedicated_resources\": {\n", + " \"machine_spec\": {\"machine_type\": \"n1-standard-4\", \"accelerator_count\": 0},\n", + " \"starting_replica_count\": 1,\n", + " \"max_replica_count\": 1,\n", + " },\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateBatchPredictionJobRequest(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "nj-dxOHen5IT" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"batchPredictionJob\": {\n", + " \"displayName\": \"traffic_20210310013516\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/6125898247828406272\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210310013516/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"modelParameters\": {\n", + " \"maxPredictions\": 2.0,\n", + " \"confidenceThreshold\": 0.5\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210310013516/batch_output/\"\n", + " }\n", + " },\n", + " \"dedicatedResources\": {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-4\"\n", + " },\n", + " \"startingReplicaCount\": 1,\n", + " \"maxReplicaCount\": 1\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "nLnR65Omn5IT" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].create_batch_prediction_job(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "86xo791tn5IU" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "eHpRHUQNn5IU" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "UJnCkCcfn5IU" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/batchPredictionJobs/6806214470445039616\",\n", + " \"displayName\": \"traffic_20210310013516\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/6125898247828406272\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210310013516/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"modelParameters\": {\n", + " \"maxPredictions\": 2.0,\n", + " \"confidenceThreshold\": 0.5\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210310013516/batch_output/\"\n", + " }\n", + " },\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"completionStats\": {\n", + " \"incompleteCount\": \"-1\"\n", + " },\n", + " \"createTime\": \"2021-03-10T14:23:59.862541Z\",\n", + " \"updateTime\": \"2021-03-10T14:23:59.862541Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batch_job_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The fully qualified ID for the batch job\n", + "batch_job_id = request.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split(\"/\")[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_get:migration,new" + }, + "source": [ + "### [projects.locations.batchPredictionJobs.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.batchPredictionJobs/get)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "OJpZjkX1n5IZ" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_get:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].get_batch_prediction_job(name=batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "h3la42lGn5Ia" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "9oovDLACn5Ia" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "PgOg5Qqln5Ia" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/batchPredictionJobs/6806214470445039616\",\n", + " \"displayName\": \"traffic_20210310013516\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/6125898247828406272\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210310013516/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"modelParameters\": {\n", + " \"maxPredictions\": 2.0,\n", + " \"confidenceThreshold\": 0.5\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210310013516/batch_output/\"\n", + " }\n", + " },\n", + " \"state\": \"JOB_STATE_RUNNING\",\n", + " \"completionStats\": {\n", + " \"incompleteCount\": \"2\"\n", + " },\n", + " \"createTime\": \"2021-03-10T14:23:59.862541Z\",\n", + " \"startTime\": \"2021-03-10T14:24:00.012555Z\",\n", + " \"updateTime\": \"2021-03-10T14:24:00.520535Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "def get_latest_predictions(gcs_out_dir):\n", + " \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n", + " folders = !gsutil ls $gcs_out_dir\n", + " latest = \"\"\n", + " for folder in folders:\n", + " subfolder = folder.split(\"/\")[-2]\n", + " if subfolder.startswith(\"prediction-\"):\n", + " if subfolder > latest:\n", + " latest = folder[:-1]\n", + " return latest\n", + "\n", + "\n", + "while True:\n", + " response = clients[\"job\"].get_batch_prediction_job(name=batch_job_id)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", response.state)\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " folder = get_latest_predictions(\n", + " response.output_config.gcs_destination.output_uri_prefix\n", + " )\n", + " ! gsutil ls $folder/prediction**\n", + "\n", + " ! gsutil cat $folder/prediction**\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Xgx0imoHn5Ib" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210310013516/batch_output/prediction-traffic_20210310013516-2021-03-10T14:23:59.650333Z/predictions_00001.jsonl\n", + "gs://migration-ucaip-trainingaip-20210310013516/batch_output/prediction-traffic_20210310013516-2021-03-10T14:23:59.650333Z/predictions_00002.jsonl\n", + "{\"instance\":{\"content\":\"gs://automl-video-demo-data/traffic_videos/highway_005.mp4\",\"mimeType\":\"video/avi\",\"timeSegmentStart\":\"0.0s\",\"timeSegmentEnd\":\"inf\"},\"prediction\":[{\"id\":\"6593913068772655104\",\"displayName\":\"pickup_suv_van\",\"timeSegmentStart\":\"35.300s\",\"timeSegmentEnd\":\"35.900s\",\"confidence\":0.5733388,\"frames\":[{\"timeOffset\":\"35.300s\",\"xMin\":0.92079705,\"xMax\":1.0,\"yMin\":0.52717,\"yMax\":0.6845663},{\"timeOffset\":\"35.400s\",\"xMin\":0.86325824,\"xMax\":0.999999,\"yMin\":0.5247129,\"yMax\":0.68186224},{\"timeOffset\":\"35.500s\",\"xMin\":0.79357177,\"xMax\":0.99050075,\"yMin\":0.5186033,\"yMax\":0.68388295},{\"timeOffset\":\"35.600s\",\"xMin\":0.7312134,\"xMax\":0.935794,\"yMin\":0.5121129,\"yMax\":0.68021643},{\"timeOffset\":\"35.700s\",\"xMin\":0.6609115,\"xMax\":0.8773811,\"yMin\":0.50215065,\"yMax\":0.6793843},{\"timeOffset\":\"35.800s\",\"xMin\":0.593415,\"xMax\":0.816827,\"yMin\":0.4967009,\"yMax\":0.677144},{\"timeOffset\":\"35.900s\",\"xMin\":0.51087815,\"xMax\":0.75138736,\"yMin\":0.4922755,\"yMax\":0.6732076}]},{\"id\":\"6593913068772655104\",\"displayName\":\"pickup_suv_van\",\"timeSegmentStart\":\"46.900s\",\"timeSegmentEnd\":\"47.900s\",\"confidence\":0.5721089,\"frames\":[{\"timeOffset\":\"46.900s\",\"xMin\":0.8938238,\"xMax\":1.0,\"yMin\":0.53446406,\"yMax\":0.6901284},{\"timeOffset\":\"47s\",\"xMin\":0.81357133,\"xMax\":0.99999857,\"yMin\":0.5293013,\"yMax\":0.6901114},{\"timeOffset\":\"47.100s\",\"xMin\":0.733154,\"xMax\":0.94289106,\"yMin\":0.52340066,\"yMax\":0.69049513},{\"timeOffset\":\"47.200s\",\"xMin\":0.6544491,\"xMax\":0.87583166,\"yMin\":0.51654726,\"yMax\":0.68932515},{\"timeOffset\":\"47.300s\",\"xMin\":0.56814355,\"xMax\":0.7984497,\"yMin\":0.50629544,\"yMax\":0.6880638},{\"timeOffset\":\"47.400s\",\"xMin\":0.47772846,\"xMax\":0.7148553,\"yMin\":0.49765483,\"yMax\":0.68850183},{\"timeOffset\":\"47.500s\",\"xMin\":0.3756373,\"xMax\":0.6187881,\"yMin\":0.49258503,\"yMax\":0.68797386},{\"timeOffset\":\"47.600s\",\"xMin\":0.28856453,\"xMax\":0.5317154,\"yMin\":0.48884195,\"yMax\":0.6842308},{\"timeOffset\":\"47.700s\",\"xMin\":0.2014918,\"xMax\":0.44464266,\"yMin\":0.48509887,\"yMax\":0.6804877},{\"timeOffset\":\"47.800s\",\"xMin\":0.11441906,\"xMax\":0.35756993,\"yMin\":0.48135576,\"yMax\":0.6767446},{\"timeOffset\":\"47.900s\",\"xMin\":0.0075410376,\"xMax\":0.16627955,\"yMin\":0.4600201,\"yMax\":0.67653906}]},{\"id\":\"6593913068772655104\",\"displayName\":\"pickup_suv_van\",\"timeSegmentStart\":\"30.700s\",\"timeSegmentEnd\":\"31.800s\",\"confidence\":0.5617838,\"frames\":[{\"timeOffset\":\"30.700s\",\"xMin\":0.9150882,\"xMax\":1.0,\"yMin\":0.5151331,\"yMax\":0.6645079},{\"timeOffset\":\"30.800s\",\"xMin\":0.86258525,\"xMax\":0.99999905,\"yMin\":0.5172026,\"yMax\":0.6603533},{\"timeOffset\":\"30.900s\",\"xMin\":0.801704,\"xMax\":0.9690958,\"yMin\":0.5114948,\"yMax\":0.65620536},{\"timeOffset\":\"31s\",\"xMin\":0.7319551,\"xMax\":0.90261525,\"yMin\":0.5059969,\"yMax\":0.65359175},{\"timeOffset\":\"31.100s\",\"xMin\":0.66124815,\"xMax\":0.83811176,\"yMin\":0.49920097,\"yMax\":0.65197486},{\"timeOffset\":\"31.200s\",\"xMin\":0.5840557,\"xMax\":0.77017546,\"yMin\":0.49158275,\"yMax\":0.6492404},{\"timeOffset\":\"31.300s\",\"xMin\":0.49507296,\"xMax\":0.7009768,\"yMin\":0.48103508,\"yMax\":0.6466039},{\"timeOffset\":\"31.400s\",\"xMin\":0.405905,\"xMax\":0.61765563,\"yMin\":0.4749309,\"yMax\":0.6409255},{\"timeOffset\":\"31.500s\",\"xMin\":0.3119387,\"xMax\":0.5226898,\"yMin\":0.46739954,\"yMax\":0.639586},{\"timeOffset\":\"31.600s\",\"xMin\":0.21487714,\"xMax\":0.4266367,\"yMin\":0.46247935,\"yMax\":0.6337805},{\"timeOffset\":\"31.700s\",\"xMin\":0.104759425,\"xMax\":0.32538062,\"yMin\":0.4524644,\"yMax\":0.63389504},{\"timeOffset\":\"31.800s\",\"xMin\":0.018637476,\"xMax\":0.18262094,\"yMin\":0.438457,\"yMax\":0.6318268}]},{\"id\":\"6593913068772655104\",\"displayName\":\"pickup_suv_van\",\"timeSegmentStart\":\"19.200s\",\"timeSegmentEnd\":\"20.400s\",\"confidence\":0.5343286,\"frames\":[{\"timeOffset\":\"19.200s\",\"xMin\":0.53855574,\"xMax\":0.7483454,\"yMin\":0.7698453,\"yMax\":0.88802785},{\"timeOffset\":\"19.300s\",\"xMin\":0.7066706,\"xMax\":0.9046804,\"yMin\":0.5262296,\"yMax\":0.68281764},{\"timeOffset\":\"19.400s\",\"xMin\":0.6643509,\"xMax\":0.8667256,\"yMin\":0.52155364,\"yMax\":0.683776},{\"timeOffset\":\"19.500s\",\"xMin\":0.60929006,\"xMax\":0.81281793,\"yMin\":0.51661617,\"yMax\":0.6828631},{\"timeOffset\":\"19.600s\",\"xMin\":0.5457967,\"xMax\":0.75832534,\"yMin\":0.51252514,\"yMax\":0.6830068},{\"timeOffset\":\"19.700s\",\"xMin\":0.48583922,\"xMax\":0.7025621,\"yMin\":0.5050785,\"yMax\":0.6833887},{\"timeOffset\":\"19.800s\",\"xMin\":0.42353436,\"xMax\":0.65267944,\"yMin\":0.499186,\"yMax\":0.68389857},{\"timeOffset\":\"19.900s\",\"xMin\":0.36298347,\"xMax\":0.5871576,\"yMin\":0.49524042,\"yMax\":0.68120617},{\"timeOffset\":\"20s\",\"xMin\":0.28549758,\"xMax\":0.51240987,\"yMin\":0.48895967,\"yMax\":0.6785187},{\"timeOffset\":\"20.100s\",\"xMin\":0.20944653,\"xMax\":0.439814,\"yMin\":0.48409462,\"yMax\":0.6736581},{\"timeOffset\":\"20.200s\",\"xMin\":0.12498511,\"xMax\":0.3544973,\"yMin\":0.47413808,\"yMax\":0.6721386},{\"timeOffset\":\"20.300s\",\"xMin\":0.047281284,\"xMax\":0.26844072,\"yMin\":0.46798372,\"yMax\":0.66819036},{\"timeOffset\":\"20.400s\",\"xMin\":-9.64848E-4,\"xMax\":0.1562836,\"yMin\":0.45929697,\"yMax\":0.66720545}]},{\"id\":\"6593913068772655104\",\"displayName\":\"pickup_suv_van\",\"timeSegmentStart\":\"21.100s\",\"timeSegmentEnd\":\"22.500s\",\"confidence\":0.5130946,\"frames\":[{\"timeOffset\":\"21.100s\",\"xMin\":0.96134996,\"xMax\":0.9994743,\"yMin\":0.5534551,\"yMax\":0.69061935},{\"timeOffset\":\"21.200s\",\"xMin\":0.70565313,\"xMax\":0.9051543,\"yMin\":0.52271163,\"yMax\":0.68131983},{\"timeOffset\":\"21.300s\",\"xMin\":0.66035485,\"xMax\":0.8695388,\"yMin\":0.5142326,\"yMax\":0.68292105},{\"timeOffset\":\"21.400s\",\"xMin\":0.60553956,\"xMax\":0.81872016,\"yMin\":0.5056492,\"yMax\":0.6834419},{\"timeOffset\":\"21.500s\",\"xMin\":0.54476696,\"xMax\":0.7676598,\"yMin\":0.5011589,\"yMax\":0.6836144},{\"timeOffset\":\"21.600s\",\"xMin\":0.48482427,\"xMax\":0.7100822,\"yMin\":0.49725318,\"yMax\":0.68293834},{\"timeOffset\":\"21.700s\",\"xMin\":0.42285097,\"xMax\":0.65375054,\"yMin\":0.49218974,\"yMax\":0.6809865},{\"timeOffset\":\"21.800s\",\"xMin\":0.35869005,\"xMax\":0.5910624,\"yMin\":0.48708224,\"yMax\":0.68020374},{\"timeOffset\":\"21.900s\",\"xMin\":0.29066974,\"xMax\":0.52481556,\"yMin\":0.4808314,\"yMax\":0.68033195},{\"timeOffset\":\"22s\",\"xMin\":0.22343048,\"xMax\":0.460608,\"yMin\":0.4752792,\"yMax\":0.68010074},{\"timeOffset\":\"22.100s\",\"xMin\":0.15115453,\"xMax\":0.39462602,\"yMin\":0.46882644,\"yMax\":0.67740893},{\"timeOffset\":\"22.200s\",\"xMin\":0.07956007,\"xMax\":0.32599604,\"yMin\":0.46620452,\"yMax\":0.6739611},{\"timeOffset\":\"22.300s\",\"xMin\":0.019409377,\"xMax\":0.20294939,\"yMin\":0.4610349,\"yMax\":0.67052174},{\"timeOffset\":\"22.400s\",\"xMin\":-0.017409453,\"xMax\":0.13916901,\"yMin\":0.45393682,\"yMax\":0.66754013},{\"timeOffset\":\"22.500s\",\"xMin\":-0.02455799,\"xMax\":0.0710371,\"yMin\":0.45522687,\"yMax\":0.66398114}]},{\"id\":\"6593913068772655104\",\"displayName\":\"pickup_suv_van\",\"timeSegmentStart\":\"7.200s\",\"timeSegmentEnd\":\"7.800s\",\"confidence\":0.50423145,\"frames\":[{\"timeOffset\":\"7.200s\",\"xMin\":0.77211607,\"xMax\":0.9202541,\"yMin\":0.55106527,\"yMax\":0.67469126},{\"timeOffset\":\"7.300s\",\"xMin\":0.9063811,\"xMax\":0.9998785,\"yMin\":0.52542484,\"yMax\":0.68584275},{\"timeOffset\":\"7.400s\",\"xMin\":0.8464132,\"xMax\":0.9988885,\"yMin\":0.5182979,\"yMax\":0.6811758},{\"timeOffset\":\"7.500s\",\"xMin\":0.7691617,\"xMax\":0.9857696,\"yMin\":0.5142189,\"yMax\":0.6805917},{\"timeOffset\":\"7.600s\",\"xMin\":0.7101507,\"xMax\":0.91866463,\"yMin\":0.50963223,\"yMax\":0.6773922},{\"timeOffset\":\"7.700s\",\"xMin\":0.6297891,\"xMax\":0.85534894,\"yMin\":0.505215,\"yMax\":0.6743725},{\"timeOffset\":\"7.800s\",\"xMin\":0.5598867,\"xMax\":0.7586217,\"yMin\":0.5045158,\"yMax\":0.67020833}]},{\"id\":\"6593913068772655104\",\"displayName\":\"pickup_suv_van\",\"timeSegmentStart\":\"60.800s\",\"timeSegmentEnd\":\"60.900s\",\"confidence\":0.5034977,\"frames\":[{\"timeOffset\":\"60.800s\",\"xMin\":0.6538362,\"xMax\":0.88029855,\"yMin\":0.59181464,\"yMax\":0.769181},{\"timeOffset\":\"60.900s\",\"xMin\":0.53221023,\"xMax\":0.7752164,\"yMin\":0.58824813,\"yMax\":0.77228796}]},{\"id\":\"6593913068772655104\",\"displayName\":\"pickup_suv_van\",\"timeSegmentStart\":\"49.800s\",\"timeSegmentEnd\":\"50s\",\"confidence\":0.5016138,\"frames\":[{\"timeOffset\":\"49.800s\",\"xMin\":0.5714027,\"xMax\":0.8481284,\"yMin\":0.63446057,\"yMax\":0.88131857},{\"timeOffset\":\"49.900s\",\"xMin\":0.3961952,\"xMax\":0.6958802,\"yMin\":0.6273483,\"yMax\":0.8836541},{\"timeOffset\":\"50s\",\"xMin\":0.20160168,\"xMax\":0.50914085,\"yMin\":0.6205848,\"yMax\":0.89831626}]}]}\n", + "{\"instance\":{\"content\":\"gs://automl-video-demo-data/traffic_videos/highway_006.mp4\",\"mimeType\":\"video/avi\",\"timeSegmentStart\":\"0.0s\",\"timeSegmentEnd\":\"inf\"},\"prediction\":[{\"id\":\"6593913068772655104\",\"displayName\":\"pickup_suv_van\",\"timeSegmentStart\":\"33.600s\",\"timeSegmentEnd\":\"34.300s\",\"confidence\":0.57430255,\"frames\":[{\"timeOffset\":\"33.600s\",\"xMin\":0.7709672,\"xMax\":0.95710844,\"yMin\":0.50214607,\"yMax\":0.663173},{\"timeOffset\":\"33.700s\",\"xMin\":0.6879625,\"xMax\":0.8816852,\"yMin\":0.5102545,\"yMax\":0.67594415},{\"timeOffset\":\"33.800s\",\"xMin\":0.60038805,\"xMax\":0.7979881,\"yMin\":0.5179747,\"yMax\":0.6842574},{\"timeOffset\":\"33.900s\",\"xMin\":0.49664575,\"xMax\":0.7060626,\"yMin\":0.5262368,\"yMax\":0.69601244},{\"timeOffset\":\"34s\",\"xMin\":0.40653434,\"xMax\":0.62025917,\"yMin\":0.5326814,\"yMax\":0.7117975},{\"timeOffset\":\"34.100s\",\"xMin\":0.29995015,\"xMax\":0.5285418,\"yMin\":0.53684795,\"yMax\":0.7238653},{\"timeOffset\":\"34.200s\",\"xMin\":0.18591899,\"xMax\":0.42276734,\"yMin\":0.5479096,\"yMax\":0.7368871},{\"timeOffset\":\"34.300s\",\"xMin\":0.06478273,\"xMax\":0.3098502,\"yMin\":0.5567089,\"yMax\":0.7530978}]},{\"id\":\"8899756077986349056\",\"displayName\":\"large_veh_bus\",\"timeSegmentStart\":\"42.800s\",\"timeSegmentEnd\":\"43.600s\",\"confidence\":0.5635853,\"frames\":[{\"timeOffset\":\"42.800s\",\"xMin\":0.896148,\"xMax\":0.99750274,\"yMin\":0.08421734,\"yMax\":0.27437803},{\"timeOffset\":\"42.900s\",\"xMin\":0.61661655,\"xMax\":0.9991679,\"yMin\":0.34556246,\"yMax\":0.7192955},{\"timeOffset\":\"43s\",\"xMin\":0.5206372,\"xMax\":0.9830786,\"yMin\":0.35251692,\"yMax\":0.72906935},{\"timeOffset\":\"43.100s\",\"xMin\":0.415886,\"xMax\":0.95419466,\"yMin\":0.3468655,\"yMax\":0.74923456},{\"timeOffset\":\"43.200s\",\"xMin\":0.33259684,\"xMax\":0.89674544,\"yMin\":0.3457283,\"yMax\":0.7659639},{\"timeOffset\":\"43.300s\",\"xMin\":0.24448554,\"xMax\":0.813682,\"yMin\":0.34896624,\"yMax\":0.77039707},{\"timeOffset\":\"43.400s\",\"xMin\":0.14297022,\"xMax\":0.7258636,\"yMin\":0.34973124,\"yMax\":0.78382397},{\"timeOffset\":\"43.500s\",\"xMin\":0.035293583,\"xMax\":0.6205815,\"yMin\":0.3503142,\"yMax\":0.797604},{\"timeOffset\":\"43.600s\",\"xMin\":-0.012641478,\"xMax\":0.5095917,\"yMin\":0.34936112,\"yMax\":0.7985512}]},{\"id\":\"6593913068772655104\",\"displayName\":\"pickup_suv_van\",\"timeSegmentStart\":\"14.300s\",\"timeSegmentEnd\":\"14.800s\",\"confidence\":0.5230016,\"frames\":[{\"timeOffset\":\"14.300s\",\"xMin\":0.6586052,\"xMax\":0.92509496,\"yMin\":0.56249046,\"yMax\":0.7750921},{\"timeOffset\":\"14.400s\",\"xMin\":0.54897916,\"xMax\":0.83694077,\"yMin\":0.57280374,\"yMax\":0.7998729},{\"timeOffset\":\"14.500s\",\"xMin\":0.42660663,\"xMax\":0.7059393,\"yMin\":0.5838014,\"yMax\":0.81943566},{\"timeOffset\":\"14.600s\",\"xMin\":0.28328148,\"xMax\":0.5759178,\"yMin\":0.5971082,\"yMax\":0.8422871},{\"timeOffset\":\"14.700s\",\"xMin\":0.14642228,\"xMax\":0.47787058,\"yMin\":0.60969377,\"yMax\":0.86163664},{\"timeOffset\":\"14.800s\",\"xMin\":0.13050511,\"xMax\":0.393794,\"yMin\":0.6079176,\"yMax\":0.8719975}]},{\"id\":\"8899756077986349056\",\"displayName\":\"large_veh_bus\",\"timeSegmentStart\":\"36.400s\",\"timeSegmentEnd\":\"37.200s\",\"confidence\":0.5050303,\"frames\":[{\"timeOffset\":\"36.400s\",\"xMin\":0.45120007,\"xMax\":0.6234123,\"yMin\":0.4991348,\"yMax\":0.68943614},{\"timeOffset\":\"36.500s\",\"xMin\":0.52608997,\"xMax\":0.97820985,\"yMin\":0.37224957,\"yMax\":0.70904154},{\"timeOffset\":\"36.600s\",\"xMin\":0.43248644,\"xMax\":0.9457052,\"yMin\":0.3685041,\"yMax\":0.7249851},{\"timeOffset\":\"36.700s\",\"xMin\":0.32448632,\"xMax\":0.8723778,\"yMin\":0.3692387,\"yMax\":0.736981},{\"timeOffset\":\"36.800s\",\"xMin\":0.24751942,\"xMax\":0.7811472,\"yMin\":0.36809424,\"yMax\":0.76570237},{\"timeOffset\":\"36.900s\",\"xMin\":0.11482327,\"xMax\":0.6963299,\"yMin\":0.36990193,\"yMax\":0.7694417},{\"timeOffset\":\"37s\",\"xMin\":0.014343479,\"xMax\":0.5851304,\"yMin\":0.37472367,\"yMax\":0.7721312},{\"timeOffset\":\"37.100s\",\"xMin\":-0.008719128,\"xMax\":0.47485036,\"yMin\":0.37370107,\"yMax\":0.77210236},{\"timeOffset\":\"37.200s\",\"xMin\":-0.0062835654,\"xMax\":0.35681728,\"yMin\":0.36186668,\"yMax\":0.811519}]},{\"id\":\"6593913068772655104\",\"displayName\":\"pickup_suv_van\",\"timeSegmentStart\":\"30.700s\",\"timeSegmentEnd\":\"31.300s\",\"confidence\":0.50222003,\"frames\":[{\"timeOffset\":\"30.700s\",\"xMin\":0.5281936,\"xMax\":0.68637884,\"yMin\":0.52747375,\"yMax\":0.68398994},{\"timeOffset\":\"30.800s\",\"xMin\":0.57842255,\"xMax\":0.78849596,\"yMin\":0.5235871,\"yMax\":0.6842843},{\"timeOffset\":\"30.900s\",\"xMin\":0.4813683,\"xMax\":0.69776064,\"yMin\":0.5304373,\"yMax\":0.69869477},{\"timeOffset\":\"31s\",\"xMin\":0.36337966,\"xMax\":0.5917747,\"yMin\":0.53825593,\"yMax\":0.7169143},{\"timeOffset\":\"31.100s\",\"xMin\":0.23924252,\"xMax\":0.4805882,\"yMin\":0.53838265,\"yMax\":0.7325624},{\"timeOffset\":\"31.200s\",\"xMin\":0.112733364,\"xMax\":0.37042505,\"yMin\":0.5456531,\"yMax\":0.7470921},{\"timeOffset\":\"31.300s\",\"xMin\":0.021284297,\"xMax\":0.18838634,\"yMin\":0.5539143,\"yMax\":0.7607572}]},{\"id\":\"6593913068772655104\",\"displayName\":\"pickup_suv_van\",\"timeSegmentStart\":\"22.400s\",\"timeSegmentEnd\":\"23s\",\"confidence\":0.5017651,\"frames\":[{\"timeOffset\":\"22.400s\",\"xMin\":0.8683299,\"xMax\":1.0,\"yMin\":0.56906176,\"yMax\":0.7567184},{\"timeOffset\":\"22.500s\",\"xMin\":0.75663006,\"xMax\":0.9971202,\"yMin\":0.5708618,\"yMax\":0.77356946},{\"timeOffset\":\"22.600s\",\"xMin\":0.65595376,\"xMax\":0.9007106,\"yMin\":0.582666,\"yMax\":0.7840566},{\"timeOffset\":\"22.700s\",\"xMin\":0.54099375,\"xMax\":0.7999944,\"yMin\":0.5906615,\"yMax\":0.80249023},{\"timeOffset\":\"22.800s\",\"xMin\":0.41166627,\"xMax\":0.68448514,\"yMin\":0.59567124,\"yMax\":0.8259323},{\"timeOffset\":\"22.900s\",\"xMin\":0.2598535,\"xMax\":0.54294544,\"yMin\":0.6135365,\"yMax\":0.84419787},{\"timeOffset\":\"23s\",\"xMin\":0.081104435,\"xMax\":0.38947436,\"yMin\":0.6304356,\"yMax\":0.8707452}]},{\"id\":\"6593913068772655104\",\"displayName\":\"pickup_suv_van\",\"timeSegmentStart\":\"45.200s\",\"timeSegmentEnd\":\"45.800s\",\"confidence\":0.5006547,\"frames\":[{\"timeOffset\":\"45.200s\",\"xMin\":0.93331534,\"xMax\":0.99947244,\"yMin\":0.5425535,\"yMax\":0.7291488},{\"timeOffset\":\"45.300s\",\"xMin\":0.85503596,\"xMax\":0.9993583,\"yMin\":0.5599785,\"yMax\":0.75284433},{\"timeOffset\":\"45.400s\",\"xMin\":0.7337758,\"xMax\":0.9910556,\"yMin\":0.5646295,\"yMax\":0.7706001},{\"timeOffset\":\"45.500s\",\"xMin\":0.6204056,\"xMax\":0.89361084,\"yMin\":0.5727049,\"yMax\":0.79307556},{\"timeOffset\":\"45.600s\",\"xMin\":0.49371505,\"xMax\":0.795267,\"yMin\":0.58835214,\"yMax\":0.81640065},{\"timeOffset\":\"45.700s\",\"xMin\":0.35610467,\"xMax\":0.6617675,\"yMin\":0.6014027,\"yMax\":0.8416757},{\"timeOffset\":\"45.800s\",\"xMin\":0.1996267,\"xMax\":0.5096757,\"yMin\":0.6206687,\"yMax\":0.8625379}]}]}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup:migration,new" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bILJnBmvn5Ib" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_model = True\n", + "delete_pipeline = True\n", + "delete_batchjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex AI fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset:\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex AI fully qualified identifier for the model\n", + "try:\n", + " if delete_model:\n", + " clients[\"model\"].delete_model(name=model_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex AI fully qualified identifier for the training pipeline\n", + "try:\n", + " if delete_pipeline:\n", + " clients[\"pipeline\"].delete_training_pipeline(name=training_pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex AI fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob:\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r gs://$BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "collapsed_sections": [], + "name": "UJ15 unified AutoML Vision Video Object Tracking.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/migration/UJ2,12 legacy Custom Training Prebuilt Container TF Keras.ipynb b/notebooks/community/migration/UJ2,12 legacy Custom Training Prebuilt Container TF Keras.ipynb new file mode 100644 index 000000000..c1d395020 --- /dev/null +++ b/notebooks/community/migration/UJ2,12 legacy Custom Training Prebuilt Container TF Keras.ipynb @@ -0,0 +1,2065 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "h5atBs3EyIsj" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_title:migration,automl,icn" + }, + "source": [ + "# Vertex SDK: Train & deploy a TensorFlow model with hosted runtimes (aka pre-built containers)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "JE-aKjayJjkF" + }, + "source": [ + "## Installation\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Iddo-8HmJjkH" + }, + "source": [ + "Install the Google *cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "xHvi6EFSJjkH" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-storage" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ktRTgB8DJjkI" + }, + "source": [ + "### Restart the Kernel\n", + "\n", + "Once you've installed the Vertex SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "x1g4mBdlJjkI" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "if not os.getenv(\"AUTORUN\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU run-time\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your GCP project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a GCP project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the AutoML APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [Google Cloud SDK](https://cloud.google.com/sdk) is already installed in AutoML Notebooks.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "jT0fijejJjkK" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported forAutoML. We recommend when possible, to choose the region closest to you. \n", + "\n", + "Currently project resources must be in the `us-central1` region to use this API." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "CTBlncfrJjkK" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "J_croPBoJjkK" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "KKxStk5bJjkL" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your GCP account\n", + "\n", + "**If you are using AutoML Notebooks**, your environment is already\n", + "authenticated. Skip this step.\n", + "\n", + "*Note: If you are on an AutoML notebook and run the cell, the cell knows to skip executing the authentication steps.*\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "6aUuFIftxXiN" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your Google Cloud account. This provides access\n", + "# to your Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on AutoML, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this tutorial in a notebook locally, replace the string\n", + " # below with the path to your service account key and run this cell to\n", + " # authenticate your Google Cloud account.\n", + " else:\n", + " %env GOOGLE_APPLICATION_CREDENTIALS your_path_to_credentials.json\n", + "\n", + " # Log in to your account on Google Cloud\n", + " ! gcloud auth login" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "MDUAZaN3JjkL" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets. " + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n", + " BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "yU6VuylGJjkM" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "QVRixM4oJjkM" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Mi-Pdh6QJjkN" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aqDptKpYJjkN" + }, + "outputs": [], + "source": [ + "! gsutil ls -al gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "fR9geV9pJjkO" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "6PsrZK5lyIsu" + }, + "outputs": [], + "source": [ + "import json\n", + "import time\n", + "\n", + "from google.protobuf.json_format import MessageToJson\n", + "from google.protobuf.struct_pb2 import Value\n", + "from googleapiclient import discovery, errors" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "lbv411XjJjkP" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Setup up the following constants for AutoML:\n", + "\n", + "- `PARENT`: The AutoM location root path for dataset, model and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Gx8Eos8PJjkP" + }, + "outputs": [], + "source": [ + "# AutoM location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "t5qygGYfyIsv" + }, + "source": [ + "## Clients\n", + "\n", + "We use the Google APIs Client Library for Python to call the Vertex Training and Prediction API without manually constructing HTTP requests." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "FUDz4l7GyIsv" + }, + "outputs": [], + "source": [ + "cloudml = discovery.build(\"ml\", \"v1\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ZX-ma7Xj_5rS" + }, + "source": [ + "## Prepare a trainer script" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "lwO1GXK0_5rS" + }, + "source": [ + "### Package assembly" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "AcJLu3w3yIsv" + }, + "outputs": [], + "source": [ + "! rm -rf cifar\n", + "! mkdir cifar\n", + "! touch cifar/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\\n", + "tag_build =\\n\\\n", + "tag_date = 0\"\n", + "! echo \"$setup_cfg\" > cifar/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\\n", + "# Requires TensorFlow Datasets\\n\\\n", + "setuptools.setup(\\n\\\n", + " install_requires=[\\n\\\n", + " 'tensorflow_datasets==1.3.0',\\n\\\n", + " ],\\n\\\n", + " packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > cifar/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\\n", + "Name: Custom Training CIFAR-10\\n\\\n", + "Version: 0.0.0\\n\\\n", + "Summary: Demonstration training script\\n\\\n", + "Home-page: www.google.com\\n\\\n", + "Author: Google\\n\\\n", + "Author-email: aferlitsch@google.com\\n\\\n", + "License: Public\\n\\\n", + "Description: Demo\\n\\\n", + "Platform: Vertex AI\"\n", + "! echo \"$pkg_info\" > cifar/PKG-INFO\n", + "\n", + "! mkdir cifar/trainer\n", + "! touch cifar/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "7agySmGx_5rS" + }, + "source": [ + "### Task.py contents" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "RRGD4U4ryIsw" + }, + "outputs": [], + "source": [ + "%%writefile cifar/trainer/task.py\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "import argparse\n", + "import os\n", + "import sys\n", + "\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default='/tmp/saved_model', type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.01, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=10, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=200, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "print('DEVICES', device_lib.list_local_devices())\n", + "\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "BUFFER_SIZE = 10000\n", + "BATCH_SIZE = 64\n", + "\n", + "def make_datasets_unbatched():\n", + " def scale(image, label):\n", + " image = tf.cast(image, tf.float32)\n", + " image /= 255.0\n", + " return image, label\n", + "\n", + " datasets, info = tfds.load(name='cifar10',\n", + " with_info=True,\n", + " as_supervised=True)\n", + " return datasets['train'].map(scale).cache().shuffle(BUFFER_SIZE).repeat()\n", + "\n", + "def build_and_compile_cnn_model():\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu', input_shape=(32, 32, 3)),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu'),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Flatten(),\n", + " tf.keras.layers.Dense(10, activation='softmax')\n", + " ])\n", + " model.compile(\n", + " loss=tf.keras.losses.sparse_categorical_crossentropy,\n", + " optimizer=tf.keras.optimizers.SGD(learning_rate=args.lr),\n", + " metrics=['accuracy'])\n", + " return model\n", + "\n", + "NUM_WORKERS = strategy.num_replicas_in_sync\n", + "GLOBAL_BATCH_SIZE = BATCH_SIZE * NUM_WORKERS\n", + "train_dataset = make_datasets_unbatched().batch(GLOBAL_BATCH_SIZE)\n", + "\n", + "with strategy.scope():\n", + " model = build_and_compile_cnn_model()\n", + "\n", + "model.fit(x=train_dataset, epochs=args.epochs, steps_per_epoch=args.steps)\n", + "model.save(args.model_dir)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "7NYujrjG_5rT" + }, + "source": [ + "### Store training script on your Cloud Storage bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "CA5_HWnEyIsz" + }, + "outputs": [], + "source": [ + "! rm -f cifar.tar cifar.tar.gz\n", + "! tar cvf cifar.tar cifar\n", + "! gzip cifar.tar\n", + "! gsutil cp cifar.tar.gz gs://$BUCKET_NAME/trainer_cifar.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_create_and_deploy_model:migration" + }, + "source": [ + "## Train a model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0oqIBOSnJjkW" + }, + "source": [ + "### [projects.jobs.create](https://cloud.google.com/ai-platform/training/docs/reference/rest/v1/projects.jobs/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "d1hFxaOKyIs2" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "FmBnuvysyIs2" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_job_TF_\" + TIMESTAMP\n", + "\n", + "TRAINING_INPUTS = {\n", + " \"scaleTier\": \"CUSTOM\",\n", + " \"masterType\": \"n1-standard-4\",\n", + " \"masterConfig\": {\"acceleratorConfig\": {\"count\": \"1\", \"type\": \"NVIDIA_TESLA_K80\"}},\n", + " \"packageUris\": [\"gs://\" + BUCKET_NAME + \"/trainer_cifar.tar.gz\"],\n", + " \"pythonModule\": \"trainer.task\",\n", + " \"args\": [\n", + " \"--model-dir=\" + \"gs://{}/{}\".format(BUCKET_NAME, JOB_NAME),\n", + " \"--epochs=\" + str(20),\n", + " \"--steps=\" + str(100),\n", + " \"--distribute=\" + \"single\",\n", + " ],\n", + " \"region\": REGION,\n", + " \"runtimeVersion\": \"2.1\",\n", + " \"pythonVersion\": \"3.7\",\n", + "}\n", + "\n", + "body = {\"jobId\": JOB_NAME, \"trainingInput\": TRAINING_INPUTS}\n", + "\n", + "request = cloudml.projects().jobs().create(parent=PARENT)\n", + "request.body = json.loads(json.dumps(body, indent=2))\n", + "\n", + "print(json.dumps(json.loads(request.to_json()), indent=2))\n", + "\n", + "request = cloudml.projects().jobs().create(parent=PARENT, body=body)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"uri\": \"https://ml.googleapis.com/v1/projects/migration-ucaip-training/jobs?alt=json\",\n", + " \"method\": \"POST\",\n", + " \"body\": {\n", + " \"jobId\": \"custom_job_TF_20210325211532\",\n", + " \"trainingInput\": {\n", + " \"scaleTier\": \"CUSTOM\",\n", + " \"masterType\": \"n1-standard-4\",\n", + " \"masterConfig\": {\n", + " \"acceleratorConfig\": {\n", + " \"count\": \"1\",\n", + " \"type\": \"NVIDIA_TESLA_K80\"\n", + " }\n", + " },\n", + " \"packageUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210325211532/trainer_cifar.tar.gz\"\n", + " ],\n", + " \"pythonModule\": \"trainer.task\",\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210325211532/custom_job_TF_20210325211532\",\n", + " \"--epochs=20\",\n", + " \"--steps=100\",\n", + " \"--distribute=single\"\n", + " ],\n", + " \"region\": \"us-central1\",\n", + " \"runtimeVersion\": \"2.1\",\n", + " \"pythonVersion\": \"3.7\"\n", + " }\n", + " },\n", + " \"headers\": {\n", + " \"accept\": \"application/json\",\n", + " \"accept-encoding\": \"gzip, deflate\",\n", + " \"user-agent\": \"(gzip)\",\n", + " \"x-goog-api-client\": \"gdcl/1.12.8 gl-python/3.7.8\"\n", + " },\n", + " \"methodId\": \"ml.projects.jobs.create\",\n", + " \"resumable\": null,\n", + " \"response_callbacks\": [],\n", + " \"_in_error_state\": false,\n", + " \"body_size\": 0,\n", + " \"resumable_uri\": null,\n", + " \"resumable_progress\": 0\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "w9N482wpyIs2" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cCL6ohxxyIs3" + }, + "outputs": [], + "source": [ + "response = request.execute()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "zV5RiSOmyIs3" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "PFIKWKcNyIs3" + }, + "outputs": [], + "source": [ + "print(json.dumps(response, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"jobId\": \"custom_job_TF_20210325211532\",\n", + " \"trainingInput\": {\n", + " \"scaleTier\": \"CUSTOM\",\n", + " \"masterType\": \"n1-standard-4\",\n", + " \"packageUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210325211532/trainer_cifar.tar.gz\"\n", + " ],\n", + " \"pythonModule\": \"trainer.task\",\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210325211532/custom_job_TF_20210325211532\",\n", + " \"--epochs=20\",\n", + " \"--steps=100\",\n", + " \"--distribute=single\"\n", + " ],\n", + " \"region\": \"us-central1\",\n", + " \"runtimeVersion\": \"2.1\",\n", + " \"pythonVersion\": \"3.7\",\n", + " \"masterConfig\": {\n", + " \"acceleratorConfig\": {\n", + " \"count\": \"1\",\n", + " \"type\": \"NVIDIA_TESLA_K80\"\n", + " }\n", + " }\n", + " },\n", + " \"createTime\": \"2021-03-25T21:15:40Z\",\n", + " \"state\": \"QUEUED\",\n", + " \"trainingOutput\": {},\n", + " \"etag\": \"dH4whflp8Fg=\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "training_pipeline_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom training job\n", + "custom_training_id = f'{PARENT}/jobs/{response[\"jobId\"]}'\n", + "# The short numeric ID for the custom training job\n", + "custom_training_short_id = response[\"jobId\"]\n", + "\n", + "print(custom_training_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Pq4osl_GyIs3" + }, + "source": [ + "### [projects.jobs.get](https://cloud.google.com/ai-platform/training/docs/reference/rest/v1/projects.jobs/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "JIENR5rRyIs3" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "DWPhOIMvyIs3" + }, + "outputs": [], + "source": [ + "request = cloudml.projects().jobs().get(name=custom_training_id)\n", + "\n", + "response = request.execute()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "P6-XV2JUyIs4" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pmm7CEQayIs4" + }, + "outputs": [], + "source": [ + "print(json.dumps(response, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"jobId\": \"custom_job_TF_20210325211532\",\n", + " \"trainingInput\": {\n", + " \"scaleTier\": \"CUSTOM\",\n", + " \"masterType\": \"n1-standard-4\",\n", + " \"packageUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210325211532/trainer_cifar.tar.gz\"\n", + " ],\n", + " \"pythonModule\": \"trainer.task\",\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210325211532/custom_job_TF_20210325211532\",\n", + " \"--epochs=20\",\n", + " \"--steps=100\",\n", + " \"--distribute=single\"\n", + " ],\n", + " \"region\": \"us-central1\",\n", + " \"runtimeVersion\": \"2.1\",\n", + " \"pythonVersion\": \"3.7\",\n", + " \"masterConfig\": {\n", + " \"acceleratorConfig\": {\n", + " \"count\": \"1\",\n", + " \"type\": \"NVIDIA_TESLA_K80\"\n", + " }\n", + " }\n", + " },\n", + " \"createTime\": \"2021-03-25T21:15:40Z\",\n", + " \"state\": \"PREPARING\",\n", + " \"trainingOutput\": {},\n", + " \"etag\": \"eLnYfClHtKU=\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = cloudml.projects().jobs().get(name=custom_training_id).execute()\n", + " if response[\"state\"] != \"SUCCEEDED\":\n", + " print(\"Training job has not completed:\", response[\"state\"])\n", + " if response[\"state\"] == \"FAILED\":\n", + " break\n", + " else:\n", + " break\n", + " time.sleep(20)\n", + "\n", + "# model artifact output directory on Google Cloud Storage\n", + "model_artifact_dir = response[\"trainingInput\"][\"args\"][0].split(\"=\")[-1]\n", + "print(\"artifact location \" + model_artifact_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bQ5FRbQtyIs4" + }, + "source": [ + "### Serving function for trained model (image data)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "6GpykrtAyIs4" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(model_artifact_dir)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Hj6KWo31yIs4" + }, + "outputs": [], + "source": [ + "CONCRETE_INPUT = \"numpy_inputs\"\n", + "\n", + "\n", + "def _preprocess(bytes_input):\n", + " decoded = tf.io.decode_jpeg(bytes_input, channels=3)\n", + " decoded = tf.image.convert_image_dtype(decoded, tf.float32)\n", + " resized = tf.image.resize(decoded, size=(32, 32))\n", + " rescale = tf.cast(resized / 255.0, tf.float32)\n", + " return rescale\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def preprocess_fn(bytes_inputs):\n", + " decoded_images = tf.map_fn(\n", + " _preprocess, bytes_inputs, dtype=tf.float32, back_prop=False\n", + " )\n", + " return {\n", + " CONCRETE_INPUT: decoded_images\n", + " } # User needs to make sure the key matches model's input\n", + "\n", + "\n", + "m_call = tf.function(model.call).get_concrete_function(\n", + " [tf.TensorSpec(shape=[None, 32, 32, 3], dtype=tf.float32, name=CONCRETE_INPUT)]\n", + ")\n", + "\n", + "\n", + "@tf.function(\n", + " input_signature=[tf.TensorSpec([None], tf.string), tf.TensorSpec([None], tf.string)]\n", + ")\n", + "def serving_fn(bytes_inputs, key):\n", + " images = preprocess_fn(bytes_inputs)\n", + " prob = m_call(**images)\n", + " return {\"prediction\": prob, \"key\": key}\n", + "\n", + "\n", + "tf.saved_model.save(\n", + " model,\n", + " model_artifact_dir,\n", + " signatures={\n", + " \"serving_default\": serving_fn,\n", + " },\n", + ")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wlu4ZnUKyIs5" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_artifact_dir)\n", + "\n", + "tensors_specs = list(loaded.signatures[\"serving_default\"].structured_input_signature)\n", + "print(\"Tensors specs:\", tensors_specs)\n", + "\n", + "input_name = [v for k, v in tensors_specs[1].items() if k != \"key\"][0].name\n", + "print(\"Bytes input tensor name:\", input_name)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "Tensors specs: [(), {'bytes_inputs': TensorSpec(shape=(None,), dtype=tf.string, name='bytes_inputs'), 'key': TensorSpec(shape=(None,), dtype=tf.string, name='key')}]\n", + "Bytes input tensor name: bytes_inputs\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "a7H7jhulyIs7" + }, + "source": [ + "## Make batch predictions" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "fEUvTTuTyIs7" + }, + "source": [ + "### Prepare files for batch prediction" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "HOOHm0SpyIs7" + }, + "outputs": [], + "source": [ + "import base64\n", + "import json\n", + "\n", + "import cv2\n", + "import numpy as np\n", + "import tensorflow as tf\n", + "\n", + "(_, _), (x_test, y_test) = tf.keras.datasets.cifar10.load_data()\n", + "\n", + "test_image_1, test_label_1 = x_test[0], y_test[0]\n", + "test_image_2, test_label_2 = x_test[1], y_test[1]\n", + "\n", + "cv2.imwrite(\"tmp1.jpg\", (test_image_1 * 255).astype(np.uint8))\n", + "cv2.imwrite(\"tmp2.jpg\", (test_image_2 * 255).astype(np.uint8))\n", + "\n", + "gcs_input_uri = \"gs://\" + BUCKET_NAME + \"/\" + \"test.json\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " for img in [\"tmp1.jpg\", \"tmp2.jpg\"]:\n", + " bytes = tf.io.read_file(img)\n", + " b64str = base64.b64encode(bytes.numpy()).decode(\"utf-8\")\n", + " f.write(json.dumps({\"key\": img, input_name: {\"b64\": b64str}}) + \"\\n\")\n", + "\n", + "! gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\"key\": \"tmp1.jpg\", \"bytes_inputs\": {\"b64\": \"/9j/4AAQSkZJRgABAQAAAQABAAD/2wBDAAIBAQEBAQIBAQECAgICAgQDAgICAgUEBAMEBgUGBgYFBgYGBwkIBgcJBwYGCAsICQoKCgoKBggLDAsKDAkKCgr/2wBDAQICAgICAgUDAwUKBwYHCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgr/wAARCAAgACADASIAAhEBAxEB/8QAHwAAAQUBAQEBAQEAAAAAAAAAAAECAwQFBgcICQoL/8QAtRAAAgEDAwIEAwUFBAQAAAF9AQIDAAQRBRIhMUEGE1FhByJxFDKBkaEII0KxwRVS0fAkM2JyggkKFhcYGRolJicoKSo0NTY3ODk6Q0RFRkdISUpTVFVWV1hZWmNkZWZnaGlqc3R1dnd4eXqDhIWGh4iJipKTlJWWl5iZmqKjpKWmp6ipqrKztLW2t7i5usLDxMXGx8jJytLT1NXW19jZ2uHi4+Tl5ufo6erx8vP09fb3+Pn6/8QAHwEAAwEBAQEBAQEBAQAAAAAAAAECAwQFBgcICQoL/8QAtREAAgECBAQDBAcFBAQAAQJ3AAECAxEEBSExBhJBUQdhcRMiMoEIFEKRobHBCSMzUvAVYnLRChYkNOEl8RcYGRomJygpKjU2Nzg5OkNERUZHSElKU1RVVldYWVpjZGVmZ2hpanN0dXZ3eHl6goOEhYaHiImKkpOUlZaXmJmaoqOkpaanqKmqsrO0tba3uLm6wsPExcbHyMnK0tPU1dbX2Nna4uPk5ebn6Onq8vP09fb3+Pn6/9oADAMBAAIRAxEAPwD6E1zw/qemaZY669mkdtqsTPZTpMH85Y3KMcKeOR36444NZGj2/ibWPHaaPeHSLXRbq3jSw1O7u3V3u9zb0ZAh+QIFO4EkliCBjnwv9lfxtrviTxBbW974le/0nQ/h5ohms7m4b92bhVlkEfPIDuwJ6gyADgCuWh1fxP8As6/tGad8H5PiRrHjW6tNd1O/iXUr5Z7mx0uSZlinHODiRQCqrgGTGPmwPyqfClGlnM6Em3TSi/N3Wtnto015H6y+MK08kp14QSqScle6tFxel0+6aZ9d6/rvhXwH4407wWtq+uSXth9pa5jcwKUBIbyxzkL0Ock8nHQV2x0NtN0Gw8a6PDOunXc3liO5GZIGxwG6YBxx1x0zkV4L8Xfij4k8X/Gr4V+HdJtDpdgui3GoajJBAXlkuGvNoUEDcD5MYyuN3zEnpX0B4Q+Iunafdap8OPFCG/sL+PzLkGNgbQB1O7Jxh1JOCOvHXNfUYrh/LPqMo0oKDgvdl10117nzGD4izR5hGdWcp8zs4+umisflx8DNXi/Z/wDHviPTfiP4g+x2WieFtV03U5r9miLw2ilonTIySWijZCB6Yr2X4R/tQT/tC/s56f8AGn4C/AvxTrXiq7jksW1G78NxRlNiRxIrzO5EwiVHAePAfeoO1lIrqv2pf2Xz+1t+z3feC9E1GLSvE2paQtraa1cISXiEqu9tKVydrbMZ5Kkg8jIr234a/Bq7+EngjQPAng3wzB/ZOl6ZFa2tpp/yeWiqFB2Hq2ASeuTz15r9ixHBa+vSp1JXpxXuy6vyfpbXuz8jocUyWCVSirTb1j09V95e+E3hnwXr8dn8QPjLaSWZBguP+EcudKSW6gnSMfLHOrcQh2djCSAxY5BxkzfEDx1H4n8ZyvpEC2WnMAwighMe8hvl3gZyQCB15K5xWNq3iKbVNVk8MW91NZzxLllkt9jL2z0I/DrXCeG47T4seNL3wN4c1nULKPTY2GoX8YYNcSkfKisxwis2ASMnk9AK7f8AiHuQ47CulWlKzfM7S5W+vRfgZQ47zvA4qNako3irK8eZLpfVn//Z\"}}\n", + "{\"key\": \"tmp2.jpg\", \"bytes_inputs\": {\"b64\": \"/9j/4AAQSkZJRgABAQAAAQABAAD/2wBDAAIBAQEBAQIBAQECAgICAgQDAgICAgUEBAMEBgUGBgYFBgYGBwkIBgcJBwYGCAsICQoKCgoKBggLDAsKDAkKCgr/2wBDAQICAgICAgUDAwUKBwYHCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgr/wAARCAAgACADASIAAhEBAxEB/8QAHwAAAQUBAQEBAQEAAAAAAAAAAAECAwQFBgcICQoL/8QAtRAAAgEDAwIEAwUFBAQAAAF9AQIDAAQRBRIhMUEGE1FhByJxFDKBkaEII0KxwRVS0fAkM2JyggkKFhcYGRolJicoKSo0NTY3ODk6Q0RFRkdISUpTVFVWV1hZWmNkZWZnaGlqc3R1dnd4eXqDhIWGh4iJipKTlJWWl5iZmqKjpKWmp6ipqrKztLW2t7i5usLDxMXGx8jJytLT1NXW19jZ2uHi4+Tl5ufo6erx8vP09fb3+Pn6/8QAHwEAAwEBAQEBAQEBAQAAAAAAAAECAwQFBgcICQoL/8QAtREAAgECBAQDBAcFBAQAAQJ3AAECAxEEBSExBhJBUQdhcRMiMoEIFEKRobHBCSMzUvAVYnLRChYkNOEl8RcYGRomJygpKjU2Nzg5OkNERUZHSElKU1RVVldYWVpjZGVmZ2hpanN0dXZ3eHl6goOEhYaHiImKkpOUlZaXmJmaoqOkpaanqKmqsrO0tba3uLm6wsPExcbHyMnK0tPU1dbX2Nna4uPk5ebn6Onq8vP09fb3+Pn6/9oADAMBAAIRAxEAPwD8V5UubRQlxlSvDAtyD6dadbW91fK8lrFI6o6KzrnALHCj8cH8jX3J+1V+wR8adOjsrDxR8EPhzohsg13qKfD+zddRWBF2u0sR42AnIzjJAAzzXmnwh/Yk+D3jX4Q6h478cftgaX4Al/tR4f8AhHdf0eRruVI+IpdkbFiWLsAqgnrXZLBVFWcI6/gc0MVSlSU2eZaX+zdr954Nv/EEt7FNeWyrJHZ2moRn93tYsTuwcg7OBz19q8sa7AUEMf8AvqvoHwX+yz8Vb74gXtn4M+Euq/EbSYpV+y6vf2txptrMOAz+XIysR0xu9M4qf9pn9mf4jJoNprJ+BGgeCn0mHZfQ2OqRl793fAZUDkkAbcd8k1pUw1OUE6e/bf8AEVOs1JqT3P19/aT/AOCMf7RH7Qfx5134zeNf2z7S18Q+PkSWWDSb6406BrSMFYrWNCCAsakDbnOSSeTXg+sf8G3viHwt49ez1jxdY6zqds1veTwT+MzBdqJWnWCYb0DhXe3n2sOGMD4J2HH7IfD3xnc/EPwl4Y8R6t458M28y+EL1NRh1nS3vGXV3a1+w3S4mjCwxxpdCaFSjTNLGRImwk+A6f8AAL9oH4gaX4+tf+Ckn7Vfw4+I2k3fiW6m+HOneFNPn0WDw9piTLLbuUiYGWZsCNYp/tMtqiSbL+b7RMrqvWxVDKamZ89BOg03Q9+deupOpBRotU1CM4OMak/aSUIxkouTbUjmllc0qic60XrGNldX/dtNr/n2+aS5r3XI3ytKz+Jof+CN2r6LYHU/ibqOo2iQzFmmn8eXLfugMbDhwMcdeprg/iV+zX+zx8O9Mu9f8NaRplw9oSr6g0sl0BgdBNMzZ+i9K+svi9P+yv8ADAnRfhl4MfxNdhSDe63fzS2sJHdYpHbfjtu/KvhL9ub4tarruhy2JvJMsdjJFGFj28gKqrgKo9B6VhlvEGMzfDxm8M6N+kpRlJeT5dE/mwoZDiMO+evVb8j/2Q==\"}}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "7YXC_0bCyIs8" + }, + "source": [ + "### [projects.jobs.create](https://cloud.google.com/ai-platform/prediction/docs/reference/rest/v1/projects.jobs/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pAGChwNCyIs8" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "rbUvA_wByIs8" + }, + "outputs": [], + "source": [ + "body = {\n", + " \"jobId\": \"custom_job_TF_pred_\" + TIMESTAMP,\n", + " \"prediction_input\": {\n", + " \"input_paths\": gcs_input_uri,\n", + " \"output_path\": \"gs://\" + f\"{BUCKET_NAME}/batch_output/\",\n", + " \"data_format\": \"JSON\",\n", + " \"runtime_version\": \"2.1\",\n", + " \"uri\": model_artifact_dir,\n", + " \"region\": \"us-central1\",\n", + " },\n", + "}\n", + "\n", + "request = (\n", + " cloudml.projects()\n", + " .jobs()\n", + " .create(\n", + " parent=PARENT,\n", + " )\n", + ")\n", + "request.body = json.loads(json.dumps(body, indent=2))\n", + "\n", + "print(json.dumps(json.loads(request.to_json()), indent=2))\n", + "\n", + "request = cloudml.projects().jobs().create(parent=PARENT, body=body)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"uri\": \"https://ml.googleapis.com/v1/projects/migration-ucaip-training/jobs?alt=json\",\n", + " \"method\": \"POST\",\n", + " \"body\": {\n", + " \"jobId\": \"custom_job_TF_pred_20210325211532\",\n", + " \"prediction_input\": {\n", + " \"input_paths\": \"gs://migration-ucaip-trainingaip-20210325211532/test.json\",\n", + " \"output_path\": \"gs://migration-ucaip-trainingaip-20210325211532/batch_output/\",\n", + " \"data_format\": \"JSON\",\n", + " \"runtime_version\": \"2.1\",\n", + " \"uri\": \"gs://migration-ucaip-trainingaip-20210325211532/custom_job_TF_20210325211532\",\n", + " \"region\": \"us-central1\"\n", + " }\n", + " },\n", + " \"headers\": {\n", + " \"accept\": \"application/json\",\n", + " \"accept-encoding\": \"gzip, deflate\",\n", + " \"user-agent\": \"(gzip)\",\n", + " \"x-goog-api-client\": \"gdcl/1.12.8 gl-python/3.7.8\"\n", + " },\n", + " \"methodId\": \"ml.projects.jobs.create\",\n", + " \"resumable\": null,\n", + " \"response_callbacks\": [],\n", + " \"_in_error_state\": false,\n", + " \"body_size\": 0,\n", + " \"resumable_uri\": null,\n", + " \"resumable_progress\": 0\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "8yue_3fjyIs8" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "zmiM9YC1yIs8" + }, + "outputs": [], + "source": [ + "response = request.execute()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "itggY5N6yIs8" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gOstdZbnyIs9" + }, + "outputs": [], + "source": [ + "print(json.dumps(response, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"jobId\": \"custom_job_TF_pred_20210325211532\",\n", + " \"predictionInput\": {\n", + " \"dataFormat\": \"JSON\",\n", + " \"inputPaths\": [\n", + " \"gs://migration-ucaip-trainingaip-20210325211532/test.json\"\n", + " ],\n", + " \"outputPath\": \"gs://migration-ucaip-trainingaip-20210325211532/batch_output/\",\n", + " \"region\": \"us-central1\",\n", + " \"runtimeVersion\": \"2.1\",\n", + " \"uri\": \"gs://migration-ucaip-trainingaip-20210325211532/custom_job_TF_20210325211532\",\n", + " \"framework\": \"TENSORFLOW\"\n", + " },\n", + " \"createTime\": \"2021-03-25T21:34:56Z\",\n", + " \"state\": \"QUEUED\",\n", + " \"predictionOutput\": {\n", + " \"outputPath\": \"gs://migration-ucaip-trainingaip-20210325211532/batch_output/\"\n", + " },\n", + " \"etag\": \"QwNOFOfoKdY=\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "037bd38edf14" + }, + "outputs": [], + "source": [ + "# The full unique ID for the batch prediction job\n", + "batch_job_id = PARENT + \"/jobs/\" + response[\"jobId\"]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "j_d5kInfyIs9" + }, + "source": [ + "### [projects.jobs.get](https://cloud.google.com/ai-platform/prediction/docs/reference/rest/v1/projects.jobs/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "lbnf9fXUyIs9" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "GvQTYRA2yIs9" + }, + "outputs": [], + "source": [ + "request = cloudml.projects().jobs().get(name=batch_job_id)\n", + "\n", + "response = request.execute()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "PSJvKfb5yIs9" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aOPpWjd8yIs9" + }, + "outputs": [], + "source": [ + "print(json.dumps(response, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"jobId\": \"custom_job_TF_pred_20210325211532\",\n", + " \"predictionInput\": {\n", + " \"dataFormat\": \"JSON\",\n", + " \"inputPaths\": [\n", + " \"gs://migration-ucaip-trainingaip-20210325211532/test.json\"\n", + " ],\n", + " \"outputPath\": \"gs://migration-ucaip-trainingaip-20210325211532/batch_output/\",\n", + " \"region\": \"us-central1\",\n", + " \"runtimeVersion\": \"2.1\",\n", + " \"uri\": \"gs://migration-ucaip-trainingaip-20210325211532/custom_job_TF_20210325211532\",\n", + " \"framework\": \"TENSORFLOW\"\n", + " },\n", + " \"createTime\": \"2021-03-25T21:34:56Z\",\n", + " \"state\": \"QUEUED\",\n", + " \"predictionOutput\": {\n", + " \"outputPath\": \"gs://migration-ucaip-trainingaip-20210325211532/batch_output/\"\n", + " },\n", + " \"etag\": \"NSbtn4XnbbU=\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = request = cloudml.projects().jobs().get(name=batch_job_id).execute()\n", + " if response[\"state\"] != \"SUCCEEDED\":\n", + " print(\"The job has not completed:\", response[\"state\"])\n", + " if response[\"state\"] == \"FAILED\":\n", + " break\n", + " else:\n", + " folder = response[\"predictionInput\"][\"outputPath\"][:-1]\n", + " ! gsutil ls $folder/prediction*\n", + "\n", + " ! gsutil cat $folder/prediction*\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210325211532/batch_output/prediction.errors_stats-00000-of-00001\n", + "gs://migration-ucaip-trainingaip-20210325211532/batch_output/prediction.results-00000-of-00001\n", + "{\"prediction\": [0.033321816474199295, 0.052459586411714554, 0.1548144668340683, 0.11401787400245667, 0.17382358014583588, 0.09015274047851562, 0.19865882396697998, 0.10446511209011078, 0.029874442145228386, 0.048411525785923004], \"key\": \"tmp1.jpg\"}\n", + "{\"prediction\": [0.03346974775195122, 0.05255022272467613, 0.15449963510036469, 0.11388237029314041, 0.17408262193202972, 0.08989296853542328, 0.19814379513263702, 0.10520868003368378, 0.02989153563976288, 0.04837837815284729], \"key\": \"tmp2.jpg\"}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Tvjk99efyIs9" + }, + "source": [ + "## Make online predictions" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Gi_osqnJyIs4" + }, + "source": [ + "### Deploy the model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "guF8Wj1hyIs5" + }, + "source": [ + "### [projects.models.create](https://cloud.google.com/ai-platform/prediction/docs/reference/rest/v1/projects.models/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "g-I23pXRyIs5" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "rjRGd04iyIs5" + }, + "outputs": [], + "source": [ + "request = cloudml.projects().models().create(parent=PARENT)\n", + "request.body = json.loads(json.dumps({\"name\": \"custom_job_TF_\" + TIMESTAMP}, indent=2))\n", + "\n", + "print(json.dumps(json.loads(request.to_json()), indent=2))\n", + "\n", + "request = (\n", + " cloudml.projects()\n", + " .models()\n", + " .create(parent=PARENT, body={\"name\": \"custom_job_TF_\" + TIMESTAMP})\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"uri\": \"https://ml.googleapis.com/v1/projects/migration-ucaip-training/models?alt=json\",\n", + " \"method\": \"POST\",\n", + " \"body\": {\n", + " \"name\": \"custom_job_TF_20210325211532\"\n", + " },\n", + " \"headers\": {\n", + " \"accept\": \"application/json\",\n", + " \"accept-encoding\": \"gzip, deflate\",\n", + " \"user-agent\": \"(gzip)\",\n", + " \"x-goog-api-client\": \"gdcl/1.12.8 gl-python/3.7.8\"\n", + " },\n", + " \"methodId\": \"ml.projects.models.create\",\n", + " \"resumable\": null,\n", + " \"response_callbacks\": [],\n", + " \"_in_error_state\": false,\n", + " \"body_size\": 0,\n", + " \"resumable_uri\": null,\n", + " \"resumable_progress\": 0\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "nAwWTbADyIs5" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "j0-FL4T6yIs6" + }, + "outputs": [], + "source": [ + "response = request.execute()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "9sK2H5jZyIs6" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "DU5a9qR4yIs6" + }, + "outputs": [], + "source": [ + "print(json.dumps(response, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/migration-ucaip-training/models/custom_job_TF_20210325211532\",\n", + " \"regions\": [\n", + " \"us-central1\"\n", + " ],\n", + " \"etag\": \"fFH1QQbH3tA=\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cf37425915f9" + }, + "outputs": [], + "source": [ + "# The full unique ID for the training pipeline\n", + "model_id = response[\"name\"]\n", + "# The short numeric ID for the training pipeline\n", + "model_short_name = model_id.split(\"/\")[-1]\n", + "\n", + "print(model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "vJrwNIxJyIs6" + }, + "source": [ + "### [projects.models.versions.create](https://cloud.google.com/ai-platform/prediction/docs/reference/rest/v1/projects.models.versions/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "SFIKWys2yIs6" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cD2V3bjhyIs6" + }, + "outputs": [], + "source": [ + "version = {\n", + " \"name\": \"custom_job_TF_\" + TIMESTAMP,\n", + " \"deploymentUri\": model_artifact_dir,\n", + " \"runtimeVersion\": \"2.1\",\n", + " \"framework\": \"TENSORFLOW\",\n", + " \"pythonVersion\": \"3.7\",\n", + " \"machineType\": \"mls1-c1-m2\",\n", + "}\n", + "\n", + "request = (\n", + " cloudml.projects()\n", + " .models()\n", + " .versions()\n", + " .create(\n", + " parent=model_id,\n", + " )\n", + ")\n", + "request.body = json.loads(json.dumps(version, indent=2))\n", + "\n", + "print(json.dumps(json.loads(request.to_json()), indent=2))\n", + "\n", + "request = cloudml.projects().models().versions().create(parent=model_id, body=version)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"uri\": \"https://ml.googleapis.com/v1/projects/migration-ucaip-training/models/custom_job_TF_20210325211532/versions?alt=json\",\n", + " \"method\": \"POST\",\n", + " \"body\": {\n", + " \"name\": \"custom_job_TF_20210325211532\",\n", + " \"deploymentUri\": \"gs://migration-ucaip-trainingaip-20210325211532/custom_job_TF_20210325211532\",\n", + " \"runtimeVersion\": \"2.1\",\n", + " \"framework\": \"TENSORFLOW\",\n", + " \"pythonVersion\": \"3.7\",\n", + " \"machineType\": \"mls1-c1-m2\"\n", + " },\n", + " \"headers\": {\n", + " \"accept\": \"application/json\",\n", + " \"accept-encoding\": \"gzip, deflate\",\n", + " \"user-agent\": \"(gzip)\",\n", + " \"x-goog-api-client\": \"gdcl/1.12.8 gl-python/3.7.8\"\n", + " },\n", + " \"methodId\": \"ml.projects.models.versions.create\",\n", + " \"resumable\": null,\n", + " \"response_callbacks\": [],\n", + " \"_in_error_state\": false,\n", + " \"body_size\": 0,\n", + " \"resumable_uri\": null,\n", + " \"resumable_progress\": 0\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "TGZSWzgQyIs6" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "GS5at2kdyIs7" + }, + "outputs": [], + "source": [ + "response = request.execute()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "PPzFKPcayIs7" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ql_RUaDGyIs7" + }, + "outputs": [], + "source": [ + "print(json.dumps(response, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/migration-ucaip-training/operations/create_custom_job_TF_20210325211532_custom_job_TF_20210325211532-1616708521927\",\n", + " \"metadata\": {\n", + " \"@type\": \"type.googleapis.com/google.cloud.ml.v1.OperationMetadata\",\n", + " \"createTime\": \"2021-03-25T21:42:02Z\",\n", + " \"operationType\": \"CREATE_VERSION\",\n", + " \"modelName\": \"projects/migration-ucaip-training/models/custom_job_TF_20210325211532\",\n", + " \"version\": {\n", + " \"name\": \"projects/migration-ucaip-training/models/custom_job_TF_20210325211532/versions/custom_job_TF_20210325211532\",\n", + " \"deploymentUri\": \"gs://migration-ucaip-trainingaip-20210325211532/custom_job_TF_20210325211532\",\n", + " \"createTime\": \"2021-03-25T21:42:01Z\",\n", + " \"runtimeVersion\": \"2.1\",\n", + " \"etag\": \"3vf44xGDtdw=\",\n", + " \"framework\": \"TENSORFLOW\",\n", + " \"machineType\": \"mls1-c1-m2\",\n", + " \"pythonVersion\": \"3.7\"\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "6a9e3ae658cb" + }, + "outputs": [], + "source": [ + "# The full unique ID for the model version\n", + "model_version_name = response[\"metadata\"][\"version\"][\"name\"]\n", + "\n", + "print(model_version_name)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "9b9af6d5c234" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = (\n", + " cloudml.projects().models().versions().get(name=model_version_name).execute()\n", + " )\n", + " if response[\"state\"] == \"READY\":\n", + " print(\"Model version created.\")\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sqbcNfhKyIs-" + }, + "source": [ + "### Prepare input for online prediction" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "92334e84c003" + }, + "outputs": [], + "source": [ + "import base64\n", + "import json\n", + "\n", + "import cv2\n", + "import tensorflow as tf\n", + "\n", + "(_, _), (x_test, y_test) = tf.keras.datasets.cifar10.load_data()\n", + "\n", + "test_image_1, test_label_1 = x_test[0], y_test[0]\n", + "test_image_2, test_label_2 = x_test[1], y_test[1]\n", + "\n", + "cv2.imwrite(\"tmp1.jpg\", (test_image_1 * 255).astype(np.uint8))\n", + "cv2.imwrite(\"tmp2.jpg\", (test_image_2 * 255).astype(np.uint8))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "OoqARJbMyIs-" + }, + "source": [ + "### [projects.predict](https://cloud.google.com/ai-platform/prediction/docs/reference/rest/v1/projects/predict)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Ly6XCNkbyIs-" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "b54YfVXvyIs-" + }, + "outputs": [], + "source": [ + "instances_list = []\n", + "for img in [\"tmp1.jpg\", \"tmp2.jpg\"]:\n", + " bytes = tf.io.read_file(img)\n", + " b64str = base64.b64encode(bytes.numpy()).decode(\"utf-8\")\n", + " instances_list.append({\"key\": img, input_name: {\"b64\": b64str}})\n", + "\n", + "request = cloudml.projects().predict(name=model_version_name)\n", + "request.body = json.loads(json.dumps({\"instances\": instances_list}, indent=2))\n", + "\n", + "print(json.dumps(json.loads(request.to_json()), indent=2))\n", + "\n", + "request = cloudml.projects().predict(\n", + " name=model_version_name, body={\"instances\": instances_list}\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"uri\": \"https://ml.googleapis.com/v1/projects/migration-ucaip-training/models/custom_job_TF_20210325211532/versions/custom_job_TF_20210325211532:predict?alt=json\",\n", + " \"method\": \"POST\",\n", + " \"body\": {\n", + " \"instances\": [\n", + " {\n", + " \"key\": \"tmp1.jpg\",\n", + " \"bytes_inputs\": {\n", + " \"b64\": \"/9j/4AAQSkZJRgABAQAAAQABAAD/2wBDAAIBAQEBAQIBAQECAgICAgQDAgICAgUEBAMEBgUGBgYFBgYGBwkIBgcJBwYGCAsICQoKCgoKBggLDAsKDAkKCgr/2wBDAQICAgICAgUDAwUKBwYHCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgr/wAARCAAgACADASIAAhEBAxEB/8QAHwAAAQUBAQEBAQEAAAAAAAAAAAECAwQFBgcICQoL/8QAtRAAAgEDAwIEAwUFBAQAAAF9AQIDAAQRBRIhMUEGE1FhByJxFDKBkaEII0KxwRVS0fAkM2JyggkKFhcYGRolJicoKSo0NTY3ODk6Q0RFRkdISUpTVFVWV1hZWmNkZWZnaGlqc3R1dnd4eXqDhIWGh4iJipKTlJWWl5iZmqKjpKWmp6ipqrKztLW2t7i5usLDxMXGx8jJytLT1NXW19jZ2uHi4+Tl5ufo6erx8vP09fb3+Pn6/8QAHwEAAwEBAQEBAQEBAQAAAAAAAAECAwQFBgcICQoL/8QAtREAAgECBAQDBAcFBAQAAQJ3AAECAxEEBSExBhJBUQdhcRMiMoEIFEKRobHBCSMzUvAVYnLRChYkNOEl8RcYGRomJygpKjU2Nzg5OkNERUZHSElKU1RVVldYWVpjZGVmZ2hpanN0dXZ3eHl6goOEhYaHiImKkpOUlZaXmJmaoqOkpaanqKmqsrO0tba3uLm6wsPExcbHyMnK0tPU1dbX2Nna4uPk5ebn6Onq8vP09fb3+Pn6/9oADAMBAAIRAxEAPwD6E1zw/qemaZY669mkdtqsTPZTpMH85Y3KMcKeOR36444NZGj2/ibWPHaaPeHSLXRbq3jSw1O7u3V3u9zb0ZAh+QIFO4EkliCBjnwv9lfxtrviTxBbW974le/0nQ/h5ohms7m4b92bhVlkEfPIDuwJ6gyADgCuWh1fxP8As6/tGad8H5PiRrHjW6tNd1O/iXUr5Z7mx0uSZlinHODiRQCqrgGTGPmwPyqfClGlnM6Em3TSi/N3Wtnto015H6y+MK08kp14QSqScle6tFxel0+6aZ9d6/rvhXwH4407wWtq+uSXth9pa5jcwKUBIbyxzkL0Ock8nHQV2x0NtN0Gw8a6PDOunXc3liO5GZIGxwG6YBxx1x0zkV4L8Xfij4k8X/Gr4V+HdJtDpdgui3GoajJBAXlkuGvNoUEDcD5MYyuN3zEnpX0B4Q+Iunafdap8OPFCG/sL+PzLkGNgbQB1O7Jxh1JOCOvHXNfUYrh/LPqMo0oKDgvdl10117nzGD4izR5hGdWcp8zs4+umisflx8DNXi/Z/wDHviPTfiP4g+x2WieFtV03U5r9miLw2ilonTIySWijZCB6Yr2X4R/tQT/tC/s56f8AGn4C/AvxTrXiq7jksW1G78NxRlNiRxIrzO5EwiVHAePAfeoO1lIrqv2pf2Xz+1t+z3feC9E1GLSvE2paQtraa1cISXiEqu9tKVydrbMZ5Kkg8jIr234a/Bq7+EngjQPAng3wzB/ZOl6ZFa2tpp/yeWiqFB2Hq2ASeuTz15r9ixHBa+vSp1JXpxXuy6vyfpbXuz8jocUyWCVSirTb1j09V95e+E3hnwXr8dn8QPjLaSWZBguP+EcudKSW6gnSMfLHOrcQh2djCSAxY5BxkzfEDx1H4n8ZyvpEC2WnMAwighMe8hvl3gZyQCB15K5xWNq3iKbVNVk8MW91NZzxLllkt9jL2z0I/DrXCeG47T4seNL3wN4c1nULKPTY2GoX8YYNcSkfKisxwis2ASMnk9AK7f8AiHuQ47CulWlKzfM7S5W+vRfgZQ47zvA4qNako3irK8eZLpfVn//Z\"\n", + " }\n", + " },\n", + " {\n", + " \"key\": \"tmp2.jpg\",\n", + " \"bytes_inputs\": {\n", + " \"b64\": \"/9j/4AAQSkZJRgABAQAAAQABAAD/2wBDAAIBAQEBAQIBAQECAgICAgQDAgICAgUEBAMEBgUGBgYFBgYGBwkIBgcJBwYGCAsICQoKCgoKBggLDAsKDAkKCgr/2wBDAQICAgICAgUDAwUKBwYHCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgr/wAARCAAgACADASIAAhEBAxEB/8QAHwAAAQUBAQEBAQEAAAAAAAAAAAECAwQFBgcICQoL/8QAtRAAAgEDAwIEAwUFBAQAAAF9AQIDAAQRBRIhMUEGE1FhByJxFDKBkaEII0KxwRVS0fAkM2JyggkKFhcYGRolJicoKSo0NTY3ODk6Q0RFRkdISUpTVFVWV1hZWmNkZWZnaGlqc3R1dnd4eXqDhIWGh4iJipKTlJWWl5iZmqKjpKWmp6ipqrKztLW2t7i5usLDxMXGx8jJytLT1NXW19jZ2uHi4+Tl5ufo6erx8vP09fb3+Pn6/8QAHwEAAwEBAQEBAQEBAQAAAAAAAAECAwQFBgcICQoL/8QAtREAAgECBAQDBAcFBAQAAQJ3AAECAxEEBSExBhJBUQdhcRMiMoEIFEKRobHBCSMzUvAVYnLRChYkNOEl8RcYGRomJygpKjU2Nzg5OkNERUZHSElKU1RVVldYWVpjZGVmZ2hpanN0dXZ3eHl6goOEhYaHiImKkpOUlZaXmJmaoqOkpaanqKmqsrO0tba3uLm6wsPExcbHyMnK0tPU1dbX2Nna4uPk5ebn6Onq8vP09fb3+Pn6/9oADAMBAAIRAxEAPwD8V5UubRQlxlSvDAtyD6dadbW91fK8lrFI6o6KzrnALHCj8cH8jX3J+1V+wR8adOjsrDxR8EPhzohsg13qKfD+zddRWBF2u0sR42AnIzjJAAzzXmnwh/Yk+D3jX4Q6h478cftgaX4Al/tR4f8AhHdf0eRruVI+IpdkbFiWLsAqgnrXZLBVFWcI6/gc0MVSlSU2eZaX+zdr954Nv/EEt7FNeWyrJHZ2moRn93tYsTuwcg7OBz19q8sa7AUEMf8AvqvoHwX+yz8Vb74gXtn4M+Euq/EbSYpV+y6vf2txptrMOAz+XIysR0xu9M4qf9pn9mf4jJoNprJ+BGgeCn0mHZfQ2OqRl793fAZUDkkAbcd8k1pUw1OUE6e/bf8AEVOs1JqT3P19/aT/AOCMf7RH7Qfx5134zeNf2z7S18Q+PkSWWDSb6406BrSMFYrWNCCAsakDbnOSSeTXg+sf8G3viHwt49ez1jxdY6zqds1veTwT+MzBdqJWnWCYb0DhXe3n2sOGMD4J2HH7IfD3xnc/EPwl4Y8R6t458M28y+EL1NRh1nS3vGXV3a1+w3S4mjCwxxpdCaFSjTNLGRImwk+A6f8AAL9oH4gaX4+tf+Ckn7Vfw4+I2k3fiW6m+HOneFNPn0WDw9piTLLbuUiYGWZsCNYp/tMtqiSbL+b7RMrqvWxVDKamZ89BOg03Q9+deupOpBRotU1CM4OMak/aSUIxkouTbUjmllc0qic60XrGNldX/dtNr/n2+aS5r3XI3ytKz+Jof+CN2r6LYHU/ibqOo2iQzFmmn8eXLfugMbDhwMcdeprg/iV+zX+zx8O9Mu9f8NaRplw9oSr6g0sl0BgdBNMzZ+i9K+svi9P+yv8ADAnRfhl4MfxNdhSDe63fzS2sJHdYpHbfjtu/KvhL9ub4tarruhy2JvJMsdjJFGFj28gKqrgKo9B6VhlvEGMzfDxm8M6N+kpRlJeT5dE/mwoZDiMO+evVb8j/2Q==\"\n", + " }\n", + " }\n", + " ]\n", + " },\n", + " \"headers\": {\n", + " \"accept\": \"application/json\",\n", + " \"accept-encoding\": \"gzip, deflate\",\n", + " \"user-agent\": \"(gzip)\",\n", + " \"x-goog-api-client\": \"gdcl/1.12.8 gl-python/3.7.8\"\n", + " },\n", + " \"methodId\": \"ml.projects.predict\",\n", + " \"resumable\": null,\n", + " \"response_callbacks\": [],\n", + " \"_in_error_state\": false,\n", + " \"body_size\": 0,\n", + " \"resumable_uri\": null,\n", + " \"resumable_progress\": 0\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "jjeBHVKbyIs-" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "k5B1LQmQyIs_" + }, + "outputs": [], + "source": [ + "response = request.execute()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0NX1ETnIyIs_" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "xXBokAMLyIs_" + }, + "outputs": [], + "source": [ + "print(json.dumps(response, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"predictions\": [\n", + " {\n", + " \"key\": \"tmp1.jpg\",\n", + " \"prediction\": [\n", + " 0.033321816474199295,\n", + " 0.052459586411714554,\n", + " 0.1548144668340683,\n", + " 0.11401788890361786,\n", + " 0.17382356524467468,\n", + " 0.09015275537967682,\n", + " 0.19865882396697998,\n", + " 0.10446509718894958,\n", + " 0.02987445704638958,\n", + " 0.048411525785923004\n", + " ]\n", + " },\n", + " {\n", + " \"key\": \"tmp2.jpg\",\n", + " \"prediction\": [\n", + " 0.03346974775195122,\n", + " 0.052550218999385834,\n", + " 0.15449965000152588,\n", + " 0.11388237029314041,\n", + " 0.17408263683319092,\n", + " 0.08989296108484268,\n", + " 0.19814379513263702,\n", + " 0.10520866513252258,\n", + " 0.02989153563976288,\n", + " 0.04837837815284729\n", + " ]\n", + " }\n", + " ]\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "4aeaf9140e29" + }, + "source": [ + "### [projects.models.versions.delete](https://cloud.google.com/ai-platform/prediction/docs/reference/rest/v1/projects.models.versions/delete)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "e7b24444ac4d" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "8215bb333b22" + }, + "outputs": [], + "source": [ + "request = cloudml.projects().models().versions().delete(name=model_version_name)\n", + "\n", + "response = request.execute()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "response:migration" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,request" + }, + "outputs": [], + "source": [ + "print(json.dumps(response, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/migration-ucaip-training/operations/delete_custom_job_TF_20210325211532_custom_job_TF_20210325211532-1616708584436\",\n", + " \"metadata\": {\n", + " \"@type\": \"type.googleapis.com/google.cloud.ml.v1.OperationMetadata\",\n", + " \"createTime\": \"2021-03-25T21:43:04Z\",\n", + " \"operationType\": \"DELETE_VERSION\",\n", + " \"modelName\": \"projects/migration-ucaip-training/models/custom_job_TF_20210325211532\",\n", + " \"version\": {\n", + " \"name\": \"projects/migration-ucaip-training/models/custom_job_TF_20210325211532/versions/custom_job_TF_20210325211532\",\n", + " \"deploymentUri\": \"gs://migration-ucaip-trainingaip-20210325211532/custom_job_TF_20210325211532\",\n", + " \"createTime\": \"2021-03-25T21:42:01Z\",\n", + " \"runtimeVersion\": \"2.1\",\n", + " \"state\": \"READY\",\n", + " \"etag\": \"Nu2QJaCl6vw=\",\n", + " \"framework\": \"TENSORFLOW\",\n", + " \"machineType\": \"mls1-c1-m2\",\n", + " \"pythonVersion\": \"3.7\"\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bQ-VVaSxJjkd" + }, + "source": [ + "# Cleanup" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "rbFGEzYPyIs_" + }, + "outputs": [], + "source": [ + "delete_model = True\n", + "delete_bucket = True\n", + "\n", + "\n", + "# Delete the model using the Vertex AI fully qualified identifier for the model\n", + "try:\n", + " if delete_model:\n", + " cloudml.projects().models().delete(name=model_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r gs://$BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "UJ2,12 legacy Custom Training Prebuilt Container TF Keras.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/migration/UJ2,12 unified Custom Training Prebuilt Container TF Keras.ipynb b/notebooks/community/migration/UJ2,12 unified Custom Training Prebuilt Container TF Keras.ipynb new file mode 100644 index 000000000..5af0ecb4a --- /dev/null +++ b/notebooks/community/migration/UJ2,12 unified Custom Training Prebuilt Container TF Keras.ipynb @@ -0,0 +1,2253 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title:migration,new" + }, + "source": [ + "# Vertex SDK: Train & deploy a TensorFlow model with hosted runtimes (aka pre-built containers)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest (preview) version of Vertex SDK.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "7RCTGVIN_5rI" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-aiplatform --user" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the Google *cloud-storage* library as well.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "AsKFgsNV_5rJ" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-storage" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the Kernel\n", + "\n", + "Once you've installed the Vertex SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "I22lCeuV_5rJ" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "if not os.getenv(\"AUTORUN\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU run-time\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your GCP project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a GCP project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebooks.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex AI. We recommend when possible, to choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You cannot use a Multi-Regional Storage bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see [Region support for Vertex AI services](https://cloud.google.com/vertex-ai/docs/general/locations)\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "A-UGGt5N_5rM" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dWAPNQs4_5rN" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your GCP account\n", + "\n", + "**If you are using Google Cloud Notebooks**, your environment is already\n", + "authenticated. Skip this step.\n", + "\n", + "*Note: If you are on an Vertex notebook and run the cell, the cell knows to skip executing the authentication steps.*\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "nBegxspu_5rN" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your Google Cloud account. This provides access\n", + "# to your Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Vertex, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this tutorial in a notebook locally, replace the string\n", + " # below with the path to your service account key and run this cell to\n", + " # authenticate your Google Cloud account.\n", + " else:\n", + " %env GOOGLE_APPLICATION_CREDENTIALS your_path_to_credentials.json\n", + "\n", + " # Log in to your account on Google Cloud\n", + " ! gcloud auth login" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n", + " BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "N3slQeys_5rP" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "PnmUK9Si_5rP" + }, + "outputs": [], + "source": [ + "! gsutil ls -al gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip" + }, + "source": [ + "#### Import Vertex SDK\n", + "\n", + "Import the Vertex SDK into our Python environment.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "JhQ4R80M_5rQ" + }, + "outputs": [], + "source": [ + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex AI constants\n", + "\n", + "Setup up the following constants for Vertex AI:\n", + "\n", + "- `API_ENDPOINT`: The Vertex AI API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `API_PREDICT_ENDPOINT`: The Vertex AI API service endpoint for prediction.\n", + "- `PARENT`: The Vertex AI location root path for dataset, model and endpoint resources.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Ut9grfWV_5rQ" + }, + "outputs": [], + "source": [ + "# API Endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex AI location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients" + }, + "source": [ + "## Clients\n", + "\n", + "The Vertex SDK works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the server (Vertex).\n", + "\n", + "You will use several clients in this tutorial, so set them all up upfront.\n", + "\n", + "- Dataset Service for managed datasets.\n", + "- Model Service for managed models.\n", + "- Pipeline Service for training.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving. *Note*: Prediction has a different service endpoint.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "r1e3E3ua_5rR" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "clients[\"job\"] = create_job_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ZX-ma7Xj_5rS" + }, + "source": [ + "## Prepare a trainer script" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "lwO1GXK0_5rS" + }, + "source": [ + "### Package assembly" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "yZmd22dN_5rS" + }, + "outputs": [], + "source": [ + "! rm -rf cifar\n", + "! mkdir cifar\n", + "! touch cifar/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\\n", + "tag_build =\\n\\\n", + "tag_date = 0\"\n", + "! echo \"$setup_cfg\" > cifar/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\\n", + "# Requires TensorFlow Datasets\\n\\\n", + "setuptools.setup(\\n\\\n", + " install_requires=[\\n\\\n", + " 'tensorflow_datasets==1.3.0',\\n\\\n", + " ],\\n\\\n", + " packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > cifar/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\\n", + "Name: Custom Training CIFAR-10\\n\\\n", + "Version: 0.0.0\\n\\\n", + "Summary: Demonstration training script\\n\\\n", + "Home-page: www.google.com\\n\\\n", + "Author: Google\\n\\\n", + "Author-email: aferlitsch@google.com\\n\\\n", + "License: Public\\n\\\n", + "Description: Demo\\n\\\n", + "Platform: Vertex AI\"\n", + "! echo \"$pkg_info\" > cifar/PKG-INFO\n", + "\n", + "! mkdir cifar/trainer\n", + "! touch cifar/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "7agySmGx_5rS" + }, + "source": [ + "### Task.py contents" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "h4XYVUTB_5rS" + }, + "outputs": [], + "source": [ + "%%writefile cifar/trainer/task.py\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "import argparse\n", + "import os\n", + "import sys\n", + "\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default='/tmp/saved_model', type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.01, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=10, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=200, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "print('DEVICES', device_lib.list_local_devices())\n", + "\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "BUFFER_SIZE = 10000\n", + "BATCH_SIZE = 64\n", + "\n", + "def make_datasets_unbatched():\n", + " def scale(image, label):\n", + " image = tf.cast(image, tf.float32)\n", + " image /= 255.0\n", + " return image, label\n", + "\n", + " datasets, info = tfds.load(name='cifar10',\n", + " with_info=True,\n", + " as_supervised=True)\n", + " return datasets['train'].map(scale).cache().shuffle(BUFFER_SIZE).repeat()\n", + "\n", + "def build_and_compile_cnn_model():\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu', input_shape=(32, 32, 3)),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu'),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Flatten(),\n", + " tf.keras.layers.Dense(10, activation='softmax')\n", + " ])\n", + " model.compile(\n", + " loss=tf.keras.losses.sparse_categorical_crossentropy,\n", + " optimizer=tf.keras.optimizers.SGD(learning_rate=args.lr),\n", + " metrics=['accuracy'])\n", + " return model\n", + "\n", + "NUM_WORKERS = strategy.num_replicas_in_sync\n", + "GLOBAL_BATCH_SIZE = BATCH_SIZE * NUM_WORKERS\n", + "train_dataset = make_datasets_unbatched().batch(GLOBAL_BATCH_SIZE)\n", + "\n", + "with strategy.scope():\n", + " model = build_and_compile_cnn_model()\n", + "\n", + "model.fit(x=train_dataset, epochs=args.epochs, steps_per_epoch=args.steps)\n", + "model.save(args.model_dir)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "7NYujrjG_5rT" + }, + "source": [ + "### Store training script on your Cloud Storage bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "7vc09ylR_5rT" + }, + "outputs": [], + "source": [ + "! rm -f cifar.tar cifar.tar.gz\n", + "! tar cvf cifar.tar cifar\n", + "! gzip cifar.tar\n", + "! gsutil cp cifar.tar.gz gs://$BUCKET_NAME/trainer_cifar.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_create_and_deploy_model:migration" + }, + "source": [ + "## Train a model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0oqIBOSnJjkW" + }, + "source": [ + "### [projects.locations.customJobs.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.trainingPipelines/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "_-NJrAth_5rU" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "B2TRiVhq_5rU" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_job_TF_\" + TIMESTAMP\n", + "\n", + "TRAIN_IMAGE = \"gcr.io/cloud-aiplatform/training/tf-gpu.2-1:latest\"\n", + "TRAIN_NGPU = 1\n", + "TRAIN_GPU = aip.AcceleratorType.NVIDIA_TESLA_K80\n", + "\n", + "worker_pool_specs = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": {\n", + " \"machine_type\": \"n1-standard-4\",\n", + " \"accelerator_type\": TRAIN_GPU,\n", + " \"accelerator_count\": TRAIN_NGPU,\n", + " },\n", + " \"python_package_spec\": {\n", + " \"executor_image_uri\": TRAIN_IMAGE,\n", + " \"package_uris\": [\"gs://\" + BUCKET_NAME + \"/trainer_cifar.tar.gz\"],\n", + " \"python_module\": \"trainer.task\",\n", + " \"args\": [\n", + " \"--model-dir=\" + \"gs://{}/{}\".format(BUCKET_NAME, JOB_NAME),\n", + " \"--epochs=\" + str(20),\n", + " \"--steps=\" + str(100),\n", + " \"--distribute=\" + \"single\",\n", + " ],\n", + " },\n", + " }\n", + "]\n", + "\n", + "training_job = {\n", + " \"display_name\": JOB_NAME,\n", + " \"job_spec\": {\"worker_pool_specs\": worker_pool_specs},\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateCustomJobRequest(parent=PARENT, custom_job=training_job).__dict__[\n", + " \"_pb\"\n", + " ]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"customJob\": {\n", + " \"displayName\": \"custom_job_TF_20210227173057\",\n", + " \"jobSpec\": {\n", + " \"workerPoolSpecs\": [\n", + " {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-4\",\n", + " \"acceleratorType\": \"NVIDIA_TESLA_K80\",\n", + " \"acceleratorCount\": 1\n", + " },\n", + " \"replicaCount\": \"1\",\n", + " \"pythonPackageSpec\": {\n", + " \"executorImageUri\": \"gcr.io/cloud-aiplatform/training/tf-gpu.2-1:latest\",\n", + " \"packageUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210227173057/trainer_cifar.tar.gz\"\n", + " ],\n", + " \"pythonModule\": \"trainer.task\",\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210227173057/custom_job_TF_20210227173057\",\n", + " \"--epochs=20\",\n", + " \"--steps=100\",\n", + " \"--distribute=single\"\n", + " ]\n", + " }\n", + " }\n", + " ]\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "BRxBp8pz_5rU" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "HKdIlqNn_5rU" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].create_custom_job(parent=PARENT, custom_job=training_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "9NnOt5N6_5rX" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "2RvB7ep5_5rX" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "4Gvzcf0j_5rY" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/customJobs/2970106362064797696\",\n", + " \"displayName\": \"custom_job_TF_20210227173057\",\n", + " \"jobSpec\": {\n", + " \"workerPoolSpecs\": [\n", + " {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-4\",\n", + " \"acceleratorType\": \"NVIDIA_TESLA_K80\",\n", + " \"acceleratorCount\": 1\n", + " },\n", + " \"replicaCount\": \"1\",\n", + " \"diskSpec\": {\n", + " \"bootDiskType\": \"pd-ssd\",\n", + " \"bootDiskSizeGb\": 100\n", + " },\n", + " \"pythonPackageSpec\": {\n", + " \"executorImageUri\": \"gcr.io/cloud-aiplatform/training/tf-gpu.2-1:latest\",\n", + " \"packageUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210227173057/trainer_cifar.tar.gz\"\n", + " ],\n", + " \"pythonModule\": \"trainer.task\",\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210227173057/custom_job_TF_20210227173057\",\n", + " \"--epochs=20\",\n", + " \"--steps=100\",\n", + " \"--distribute=single\"\n", + " ]\n", + " }\n", + " }\n", + " ]\n", + " },\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"createTime\": \"2021-02-27T17:31:04.494716Z\",\n", + " \"updateTime\": \"2021-02-27T17:31:04.494716Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "training_pipeline_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom training job\n", + "custom_training_id = request.name\n", + "# The short numeric ID for the custom training job\n", + "custom_training_short_id = custom_training_id.split(\"/\")[-1]\n", + "\n", + "print(custom_training_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "zqfTUu82_5rY" + }, + "source": [ + "### [projects.locations.customJobs.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.trainingPipelines/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "-xTV4OhZ_5rY" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Tll5SwUs_5rZ" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].get_custom_job(name=custom_training_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "9dNrHBTK_5rZ" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "3txyKNv3_5rZ" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "BN1yzKND_5ra" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/customJobs/2970106362064797696\",\n", + " \"displayName\": \"custom_job_TF_20210227173057\",\n", + " \"jobSpec\": {\n", + " \"workerPoolSpecs\": [\n", + " {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-4\",\n", + " \"acceleratorType\": \"NVIDIA_TESLA_K80\",\n", + " \"acceleratorCount\": 1\n", + " },\n", + " \"replicaCount\": \"1\",\n", + " \"diskSpec\": {\n", + " \"bootDiskType\": \"pd-ssd\",\n", + " \"bootDiskSizeGb\": 100\n", + " },\n", + " \"pythonPackageSpec\": {\n", + " \"executorImageUri\": \"gcr.io/cloud-aiplatform/training/tf-gpu.2-1:latest\",\n", + " \"packageUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210227173057/trainer_cifar.tar.gz\"\n", + " ],\n", + " \"pythonModule\": \"trainer.task\",\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210227173057/custom_job_TF_20210227173057\",\n", + " \"--epochs=20\",\n", + " \"--steps=100\",\n", + " \"--distribute=single\"\n", + " ]\n", + " }\n", + " }\n", + " ]\n", + " },\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"createTime\": \"2021-02-27T17:31:04.494716Z\",\n", + " \"updateTime\": \"2021-02-27T17:31:04.494716Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = clients[\"job\"].get_custom_job(name=custom_training_id)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " break\n", + " else:\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(20)\n", + "\n", + "# model artifact output directory on Google Cloud Storage\n", + "model_artifact_dir = (\n", + " response.job_spec.worker_pool_specs[0].python_package_spec.args[0].split(\"=\")[-1]\n", + ")\n", + "print(\"artifact location \" + model_artifact_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wvWy83_C_5rb" + }, + "source": [ + "## Deploy the model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "1SRF-O7V_5rb" + }, + "source": [ + "### Load the saved model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "3yubnL3r_5rb" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(model_artifact_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "_KAnkwnV_5rc" + }, + "source": [ + "### Serving function for image data" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "HrAjZhPa_5rc" + }, + "outputs": [], + "source": [ + "CONCRETE_INPUT = \"numpy_inputs\"\n", + "\n", + "\n", + "def _preprocess(bytes_input):\n", + " decoded = tf.io.decode_jpeg(bytes_input, channels=3)\n", + " decoded = tf.image.convert_image_dtype(decoded, tf.float32)\n", + " resized = tf.image.resize(decoded, size=(32, 32))\n", + " rescale = tf.cast(resized / 255.0, tf.float32)\n", + " return rescale\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def preprocess_fn(bytes_inputs):\n", + " decoded_images = tf.map_fn(\n", + " _preprocess, bytes_inputs, dtype=tf.float32, back_prop=False\n", + " )\n", + " return {\n", + " CONCRETE_INPUT: decoded_images\n", + " } # User needs to make sure the key matches model's input\n", + "\n", + "\n", + "m_call = tf.function(model.call).get_concrete_function(\n", + " [tf.TensorSpec(shape=[None, 32, 32, 3], dtype=tf.float32, name=CONCRETE_INPUT)]\n", + ")\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def serving_fn(bytes_inputs):\n", + " images = preprocess_fn(bytes_inputs)\n", + " prob = m_call(**images)\n", + " return prob\n", + "\n", + "\n", + "tf.saved_model.save(\n", + " model,\n", + " model_artifact_dir,\n", + " signatures={\n", + " \"serving_default\": serving_fn,\n", + " },\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "SQdpI3ok_5rc" + }, + "source": [ + "### Get the serving function signature" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "9JxBIHY7_5rc" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_artifact_dir)\n", + "\n", + "input_name = list(\n", + " loaded.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "\n", + "print(\"Serving function input:\", input_name)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "uSLjjGyy_5rd" + }, + "source": [ + "*Example output*:\n", + "```\n", + "Serving function input: bytes_inputs\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "COwVZtxhJjkW" + }, + "source": [ + "### [projects.locations.models.upload](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.models/upload)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "QJ0VwMc1_5rd" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "mZp4UlTb_5rd" + }, + "outputs": [], + "source": [ + "model = {\n", + " \"display_name\": \"custom_job_TF\" + TIMESTAMP,\n", + " \"metadata_schema_uri\": \"\",\n", + " \"artifact_uri\": model_artifact_dir,\n", + " \"container_spec\": {\n", + " \"image_uri\": \"gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-1:latest\"\n", + " },\n", + "}\n", + "\n", + "print(MessageToJson(aip.UploadModelRequest(parent=PARENT, model=model).__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "uSLjjGyy_5rd" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"model\": {\n", + " \"displayName\": \"custom_job_TF20210227173057\",\n", + " \"containerSpec\": {\n", + " \"imageUri\": \"gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-1:latest\"\n", + " },\n", + " \"artifactUri\": \"gs://migration-ucaip-trainingaip-20210227173057/custom_job_TF_20210227173057\"\n", + " }\n", + "}\n", + "\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "CxTMT98g_5re" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "3q-pqGTN_5re" + }, + "outputs": [], + "source": [ + "request = clients[\"model\"].upload_model(parent=PARENT, model=model)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "I3Z3mGDv_5re" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "r_hEBXn2_5re" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dQak2NW3_5re" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/8844102097923211264\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "zOvD5qZK_5re" + }, + "outputs": [], + "source": [ + "# The full unique ID for the model\n", + "model_id = result.model\n", + "\n", + "print(model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_predictions:migration" + }, + "source": [ + "## Make batch predictions\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:automl,image" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "Let's now make a batch input file, which you will store in your local Cloud Storage bucket. The batch input file can be JSONL.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "_X3Dlg4X_5rf" + }, + "outputs": [], + "source": [ + "import base64\n", + "import json\n", + "\n", + "import cv2\n", + "import numpy as np\n", + "import tensorflow as tf\n", + "\n", + "(_, _), (x_test, y_test) = tf.keras.datasets.cifar10.load_data()\n", + "\n", + "\n", + "test_image_1, test_label_1 = x_test[0], y_test[0]\n", + "test_image_2, test_label_2 = x_test[1], y_test[1]\n", + "\n", + "cv2.imwrite(\"tmp1.jpg\", (test_image_1).astype(np.uint8))\n", + "cv2.imwrite(\"tmp2.jpg\", (test_image_2).astype(np.uint8))\n", + "\n", + "gcs_input_uri = \"gs://\" + BUCKET_NAME + \"/\" + \"test.jsonl\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " bytes = tf.io.read_file(\"tmp1.jpg\")\n", + " b64str = base64.b64encode(bytes.numpy()).decode(\"utf-8\")\n", + " f.write(json.dumps({input_name: {\"b64\": b64str}}) + \"\\n\")\n", + "\n", + " bytes = tf.io.read_file(\"tmp2.jpg\")\n", + " b64str = base64.b64encode(bytes.numpy()).decode(\"utf-8\")\n", + " f.write(json.dumps({input_name: {\"b64\": b64str}}) + \"\\n\")\n", + "\n", + "! gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "N-mZZ59W_5rg" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\"bytes_inputs\": {\"b64\": \"/9j/4AAQSkZJRgABAQAAAQABAAD/2wBDAAIBAQEBAQIBAQECAgICAgQDAgICAgUEBAMEBgUGBgYFBgYGBwkIBgcJBwYGCAsICQoKCgoKBggLDAsKDAkKCgr/2wBDAQICAgICAgUDAwUKBwYHCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgr/wAARCAAgACADASIAAhEBAxEB/8QAHwAAAQUBAQEBAQEAAAAAAAAAAAECAwQFBgcICQoL/8QAtRAAAgEDAwIEAwUFBAQAAAF9AQIDAAQRBRIhMUEGE1FhByJxFDKBkaEII0KxwRVS0fAkM2JyggkKFhcYGRolJicoKSo0NTY3ODk6Q0RFRkdISUpTVFVWV1hZWmNkZWZnaGlqc3R1dnd4eXqDhIWGh4iJipKTlJWWl5iZmqKjpKWmp6ipqrKztLW2t7i5usLDxMXGx8jJytLT1NXW19jZ2uHi4+Tl5ufo6erx8vP09fb3+Pn6/8QAHwEAAwEBAQEBAQEBAQAAAAAAAAECAwQFBgcICQoL/8QAtREAAgECBAQDBAcFBAQAAQJ3AAECAxEEBSExBhJBUQdhcRMiMoEIFEKRobHBCSMzUvAVYnLRChYkNOEl8RcYGRomJygpKjU2Nzg5OkNERUZHSElKU1RVVldYWVpjZGVmZ2hpanN0dXZ3eHl6goOEhYaHiImKkpOUlZaXmJmaoqOkpaanqKmqsrO0tba3uLm6wsPExcbHyMnK0tPU1dbX2Nna4uPk5ebn6Onq8vP09fb3+Pn6/9oADAMBAAIRAxEAPwD570PxBpmp6nfaEl48lzpUqpewPCU8lpEDqMsOeD26Z55Fa+s3HhnR/Aj6xZjV7rWrW4ke/wBMtLRGRLTaux1cuPnLlhtIAAUEE5490/ao8E6F4b8P3NxZeGksNW1z4h62Iby2t1/eC3ZoozJxwSiKQOhEZJ5JrqZtI8MftFfs56j8YI/hvo/gq1u9C0ywlbTbFoLa+1SOFWlgPGRmNiQzNkiPOflyf1WHFdark0K8UlUbkvJWel1vqmn5n5MuD6MM7qUJzbpxUXazvJSWtmuzTR8iaBoXirx54H1Hxo10mhx2V/8AZltpEE7ByAV8w8YLdRjAHAz1NcSNcXUtev8AwVrE0DajaQ+YZLY4jnXPJXrkjPPTPXGDXvXwi+F3hvwh8Ffip4i1a7GqX7a1b6fp0c84SKO3Wz3FiCdpHnSHDZ2/KAOtfP8A4v8Ah1qOoWul/Efwu4sL+wk8u2IkUi7JRhtwM5RgBkHpz0xXy+F4gzNY6Mqs3NTfvR6a6adj6bGcPZX/AGfKFKEYcqupemurufqP8c9Il/aA8BeHNS+HHh/7Ze634p0rUtMhsFWUJNdsFlR8HAAWWRXBPrmvGvi5+y/B+z1+0ZqHwW+PXx08LaL4VtJI75dOtPEksgfe8krskKIDCZWdCUkyU2MRuVga5X9lr9qAfsk/tCWPjTW9Ol1XwzpurtdXei27gBJTEyJcxBsDcu/OOAwBHBwa8S+JXxltPi3431/x34y8TT/2tqmpy3V1d6h8/mOzFiN46LkgDpgcdOK/HcPxo/qMalONqkn70ei816307I/Xa/C0XjXTrO8EtJdfR/cUfiz4m8aaBJefD/4NXcd4CJ7f/hI7bVXitZ4HkPzSQMvMxRUUTAEqFGCM4EPw/wDAsnhjwZEmrzte6ipKmWeYSbAV+bYTjAJBPTgNjNbOk+HYdL0qPxPcWsN5BK2FaO43q3fHUH8eld34kku/hP4LsvHPiPRtPvZNSkU6fYSFStvED8zsqjLsq5IBwOB1Jri/4iFn2BxSq0Yxulyq8eZLp1f4ms+BMkx2FlRquVm7u0uVvrbRH//Z\"}}\n", + "{\"bytes_inputs\": {\"b64\": \"/9j/4AAQSkZJRgABAQAAAQABAAD/2wBDAAIBAQEBAQIBAQECAgICAgQDAgICAgUEBAMEBgUGBgYFBgYGBwkIBgcJBwYGCAsICQoKCgoKBggLDAsKDAkKCgr/2wBDAQICAgICAgUDAwUKBwYHCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgr/wAARCAAgACADASIAAhEBAxEB/8QAHwAAAQUBAQEBAQEAAAAAAAAAAAECAwQFBgcICQoL/8QAtRAAAgEDAwIEAwUFBAQAAAF9AQIDAAQRBRIhMUEGE1FhByJxFDKBkaEII0KxwRVS0fAkM2JyggkKFhcYGRolJicoKSo0NTY3ODk6Q0RFRkdISUpTVFVWV1hZWmNkZWZnaGlqc3R1dnd4eXqDhIWGh4iJipKTlJWWl5iZmqKjpKWmp6ipqrKztLW2t7i5usLDxMXGx8jJytLT1NXW19jZ2uHi4+Tl5ufo6erx8vP09fb3+Pn6/8QAHwEAAwEBAQEBAQEBAQAAAAAAAAECAwQFBgcICQoL/8QAtREAAgECBAQDBAcFBAQAAQJ3AAECAxEEBSExBhJBUQdhcRMiMoEIFEKRobHBCSMzUvAVYnLRChYkNOEl8RcYGRomJygpKjU2Nzg5OkNERUZHSElKU1RVVldYWVpjZGVmZ2hpanN0dXZ3eHl6goOEhYaHiImKkpOUlZaXmJmaoqOkpaanqKmqsrO0tba3uLm6wsPExcbHyMnK0tPU1dbX2Nna4uPk5ebn6Onq8vP09fb3+Pn6/9oADAMBAAIRAxEAPwD9qIntrti9vhg3KkLwR69Kbc3FrYskd1LGjOjsqNjJCjLH8Mj8xXw3+yr+3v8ABbUZL2/8L/G/4ja2L0raac/xAvEbTmndtyLFKOd5AwcZwCSccV6X8Xv22/jD4K+L2n+BPA/7H+qeP4v7LSb/AISLQNYjW0ieTmWLfIoUBQiksxA6VxwxtN0VOWn4nTPC1Y1XBHpuqftI6BZ+MrDw/FZSw2dyzRyXl3p8g/eblCgbcjBG/k8dPevU1tCWIKj/AL5r5+8aftTfCqx+H9leeM/i1pXw51aWJvtWkWF1b6ldQnkqnmRqyg9c7fXGag/Zm/aY+HL69d6MPjvr/jVNWm32M19pcgSwREyVZygAJO7PbAFZ08TUjNqpt32/AdSiuVOK2PyC/Zs/4LOfs7/s+fAbQvgz4K/Ywu7rw94Bd4op9WsbfUZ1u5CGlupHBBLSMCd2MYAA4Fe0eGf+Dm/4deO9EuvDvhvSLjSWt7MpPaw+DfNiihYgNvRWK4/hyRjn3r8WvjN8MviF4C+LPiPTvhtZ6lDo8l86W6QswDID0IHUA5x7Ve/ZF1f9pX4C/Gq1+Ifw90PV7e6mgms71o7QP58EowyMrgqwJCnB9K3w+UQxleFF4hw52lzSb5Y3aXM7Juy3dtbHRRzrCu0qlKEl17/fc/W6f/gsjpGtX40z4Zadp1280IVYYPAdsv70nO8ZQnPPToK7z4a/tKftD/ETU7TQPEur6nbpdgMmnrFHak5PUwwquPq3Wvk34QwftUfE/GtfE3xmnhm0LAiy0SwhiupgezSxouzPfb+dfdv7DPwl0rQtcivhZx4Ub1eWQtJu6lmZslmPqfWnmXD+DyjESgsSq1usYyjF+a5tWvkh18+w+IXJQpJeZ//Z\"}}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_create:migration,new" + }, + "source": [ + "### [projects.locations.batchPredictionJobs.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.batchPredictionJobs/create)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "request:migration" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_create:migration,new,request,icn" + }, + "outputs": [], + "source": [ + "batch_prediction_job = aip.BatchPredictionJob(\n", + " display_name=\"custom_job_TF\" + TIMESTAMP,\n", + " model=model_id,\n", + " input_config={\n", + " \"instances_format\": \"jsonl\",\n", + " \"gcs_source\": {\"uris\": [gcs_input_uri]},\n", + " },\n", + " model_parameters=ParseDict(\n", + " {\"confidenceThreshold\": 0.5, \"maxPredictions\": 2}, Value()\n", + " ),\n", + " output_config={\n", + " \"predictions_format\": \"jsonl\",\n", + " \"gcs_destination\": {\n", + " \"output_uri_prefix\": \"gs://\" + f\"{BUCKET_NAME}/batch_output/\"\n", + " },\n", + " },\n", + " dedicated_resources={\n", + " \"machine_spec\": {\"machine_type\": \"n1-standard-2\", \"accelerator_type\": 0},\n", + " \"starting_replica_count\": 1,\n", + " \"max_replica_count\": 1,\n", + " },\n", + ")\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateBatchPredictionJobRequest(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pX4e-aNR_5rg" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"batchPredictionJob\": {\n", + " \"displayName\": \"custom_job_TF_TF20210227173057\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/8844102097923211264\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210227173057/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"modelParameters\": {\n", + " \"maxPredictions\": 10000.0,\n", + " \"confidenceThreshold\": 0.5\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210227173057/batch_output/\"\n", + " }\n", + " },\n", + " \"dedicatedResources\": {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-2\"\n", + " },\n", + " \"startingReplicaCount\": 1,\n", + " \"maxReplicaCount\": 1\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "call:migration" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].create_batch_prediction_job(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "response:migration" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,request" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_create:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/batchPredictionJobs/659759753223733248\",\n", + " \"displayName\": \"custom_job_TF_TF20210227173057\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/8844102097923211264\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210227173057/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"modelParameters\": {\n", + " \"maxPredictions\": 10000.0,\n", + " \"confidenceThreshold\": 0.5\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210227173057/batch_output/\"\n", + " }\n", + " },\n", + " \"dedicatedResources\": {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-2\"\n", + " },\n", + " \"startingReplicaCount\": 1,\n", + " \"maxReplicaCount\": 1\n", + " },\n", + " \"manualBatchTuningParameters\": {},\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"createTime\": \"2021-02-27T18:00:30.887438Z\",\n", + " \"updateTime\": \"2021-02-27T18:00:30.887438Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batch_job_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The fully qualified ID for the batch job\n", + "batch_job_id = request.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split(\"/\")[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_get:migration,new" + }, + "source": [ + "### [projects.locations.batchPredictionJobs.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.batchPredictionJobs/get)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "oW7MtyrH_5ri" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_get:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].get_batch_prediction_job(name=batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Ewx0qI1l_5ri" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "FYSggc9c_5ri" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_get:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/batchPredictionJobs/659759753223733248\",\n", + " \"displayName\": \"custom_job_TF_TF20210227173057\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/8844102097923211264\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210227173057/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"modelParameters\": {\n", + " \"confidenceThreshold\": 0.5,\n", + " \"maxPredictions\": 10000.0\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210227173057/batch_output/\"\n", + " }\n", + " },\n", + " \"dedicatedResources\": {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-2\"\n", + " },\n", + " \"startingReplicaCount\": 1,\n", + " \"maxReplicaCount\": 1\n", + " },\n", + " \"manualBatchTuningParameters\": {},\n", + " \"state\": \"JOB_STATE_RUNNING\",\n", + " \"createTime\": \"2021-02-27T18:00:30.887438Z\",\n", + " \"startTime\": \"2021-02-27T18:00:30.938444Z\",\n", + " \"updateTime\": \"2021-02-27T18:00:30.938444Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "def get_latest_predictions(gcs_out_dir):\n", + " \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n", + " folders = !gsutil ls $gcs_out_dir\n", + " latest = \"\"\n", + " for folder in folders:\n", + " subfolder = folder.split(\"/\")[-2]\n", + " if subfolder.startswith(\"prediction-\"):\n", + " if subfolder > latest:\n", + " latest = folder[:-1]\n", + " return latest\n", + "\n", + "\n", + "while True:\n", + " response = clients[\"job\"].get_batch_prediction_job(name=batch_job_id)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", response.state)\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " folder = get_latest_predictions(\n", + " response.output_config.gcs_destination.output_uri_prefix\n", + " )\n", + " ! gsutil ls $folder/prediction*\n", + "\n", + " ! gsutil cat $folder/prediction*\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_get:migration,new,wait,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210227173057/batch_output/prediction-custom_job_TF_TF20210227173057-2021_02_27T10_00_30_820Z/prediction.errors_stats-00000-of-00001\n", + "gs://migration-ucaip-trainingaip-20210227173057/batch_output/prediction-custom_job_TF_TF20210227173057-2021_02_27T10_00_30_820Z/prediction.results-00000-of-00001\n", + "{\"instance\": {\"bytes_inputs\": {\"b64\": \"/9j/4AAQSkZJRgABAQAAAQABAAD/2wBDAAIBAQEBAQIBAQECAgICAgQDAgICAgUEBAMEBgUGBgYFBgYGBwkIBgcJBwYGCAsICQoKCgoKBggLDAsKDAkKCgr/2wBDAQICAgICAgUDAwUKBwYHCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgr/wAARCAAgACADASIAAhEBAxEB/8QAHwAAAQUBAQEBAQEAAAAAAAAAAAECAwQFBgcICQoL/8QAtRAAAgEDAwIEAwUFBAQAAAF9AQIDAAQRBRIhMUEGE1FhByJxFDKBkaEII0KxwRVS0fAkM2JyggkKFhcYGRolJicoKSo0NTY3ODk6Q0RFRkdISUpTVFVWV1hZWmNkZWZnaGlqc3R1dnd4eXqDhIWGh4iJipKTlJWWl5iZmqKjpKWmp6ipqrKztLW2t7i5usLDxMXGx8jJytLT1NXW19jZ2uHi4+Tl5ufo6erx8vP09fb3+Pn6/8QAHwEAAwEBAQEBAQEBAQAAAAAAAAECAwQFBgcICQoL/8QAtREAAgECBAQDBAcFBAQAAQJ3AAECAxEEBSExBhJBUQdhcRMiMoEIFEKRobHBCSMzUvAVYnLRChYkNOEl8RcYGRomJygpKjU2Nzg5OkNERUZHSElKU1RVVldYWVpjZGVmZ2hpanN0dXZ3eHl6goOEhYaHiImKkpOUlZaXmJmaoqOkpaanqKmqsrO0tba3uLm6wsPExcbHyMnK0tPU1dbX2Nna4uPk5ebn6Onq8vP09fb3+Pn6/9oADAMBAAIRAxEAPwD570PxBpmp6nfaEl48lzpUqpewPCU8lpEDqMsOeD26Z55Fa+s3HhnR/Aj6xZjV7rWrW4ke/wBMtLRGRLTaux1cuPnLlhtIAAUEE5490/ao8E6F4b8P3NxZeGksNW1z4h62Iby2t1/eC3ZoozJxwSiKQOhEZJ5JrqZtI8MftFfs56j8YI/hvo/gq1u9C0ywlbTbFoLa+1SOFWlgPGRmNiQzNkiPOflyf1WHFdark0K8UlUbkvJWel1vqmn5n5MuD6MM7qUJzbpxUXazvJSWtmuzTR8iaBoXirx54H1Hxo10mhx2V/8AZltpEE7ByAV8w8YLdRjAHAz1NcSNcXUtev8AwVrE0DajaQ+YZLY4jnXPJXrkjPPTPXGDXvXwi+F3hvwh8Ffip4i1a7GqX7a1b6fp0c84SKO3Wz3FiCdpHnSHDZ2/KAOtfP8A4v8Ah1qOoWul/Efwu4sL+wk8u2IkUi7JRhtwM5RgBkHpz0xXy+F4gzNY6Mqs3NTfvR6a6adj6bGcPZX/AGfKFKEYcqupemurufqP8c9Il/aA8BeHNS+HHh/7Ze634p0rUtMhsFWUJNdsFlR8HAAWWRXBPrmvGvi5+y/B+z1+0ZqHwW+PXx08LaL4VtJI75dOtPEksgfe8krskKIDCZWdCUkyU2MRuVga5X9lr9qAfsk/tCWPjTW9Ol1XwzpurtdXei27gBJTEyJcxBsDcu/OOAwBHBwa8S+JXxltPi3431/x34y8TT/2tqmpy3V1d6h8/mOzFiN46LkgDpgcdOK/HcPxo/qMalONqkn70ei816307I/Xa/C0XjXTrO8EtJdfR/cUfiz4m8aaBJefD/4NXcd4CJ7f/hI7bVXitZ4HkPzSQMvMxRUUTAEqFGCM4EPw/wDAsnhjwZEmrzte6ipKmWeYSbAV+bYTjAJBPTgNjNbOk+HYdL0qPxPcWsN5BK2FaO43q3fHUH8eld34kku/hP4LsvHPiPRtPvZNSkU6fYSFStvED8zsqjLsq5IBwOB1Jri/4iFn2BxSq0Yxulyq8eZLp1f4ms+BMkx2FlRquVm7u0uVvrbRH//Z\"}}, \"prediction\": [0.0407731421, 0.125140116, 0.118551917, 0.100501947, 0.128865793, 0.089787662, 0.157575116, 0.121281914, 0.0312845968, 0.0862377882]}\n", + "{\"instance\": {\"bytes_inputs\": {\"b64\": \"/9j/4AAQSkZJRgABAQAAAQABAAD/2wBDAAIBAQEBAQIBAQECAgICAgQDAgICAgUEBAMEBgUGBgYFBgYGBwkIBgcJBwYGCAsICQoKCgoKBggLDAsKDAkKCgr/2wBDAQICAgICAgUDAwUKBwYHCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgr/wAARCAAgACADASIAAhEBAxEB/8QAHwAAAQUBAQEBAQEAAAAAAAAAAAECAwQFBgcICQoL/8QAtRAAAgEDAwIEAwUFBAQAAAF9AQIDAAQRBRIhMUEGE1FhByJxFDKBkaEII0KxwRVS0fAkM2JyggkKFhcYGRolJicoKSo0NTY3ODk6Q0RFRkdISUpTVFVWV1hZWmNkZWZnaGlqc3R1dnd4eXqDhIWGh4iJipKTlJWWl5iZmqKjpKWmp6ipqrKztLW2t7i5usLDxMXGx8jJytLT1NXW19jZ2uHi4+Tl5ufo6erx8vP09fb3+Pn6/8QAHwEAAwEBAQEBAQEBAQAAAAAAAAECAwQFBgcICQoL/8QAtREAAgECBAQDBAcFBAQAAQJ3AAECAxEEBSExBhJBUQdhcRMiMoEIFEKRobHBCSMzUvAVYnLRChYkNOEl8RcYGRomJygpKjU2Nzg5OkNERUZHSElKU1RVVldYWVpjZGVmZ2hpanN0dXZ3eHl6goOEhYaHiImKkpOUlZaXmJmaoqOkpaanqKmqsrO0tba3uLm6wsPExcbHyMnK0tPU1dbX2Nna4uPk5ebn6Onq8vP09fb3+Pn6/9oADAMBAAIRAxEAPwD9qIntrti9vhg3KkLwR69Kbc3FrYskd1LGjOjsqNjJCjLH8Mj8xXw3+yr+3v8ABbUZL2/8L/G/4ja2L0raac/xAvEbTmndtyLFKOd5AwcZwCSccV6X8Xv22/jD4K+L2n+BPA/7H+qeP4v7LSb/AISLQNYjW0ieTmWLfIoUBQiksxA6VxwxtN0VOWn4nTPC1Y1XBHpuqftI6BZ+MrDw/FZSw2dyzRyXl3p8g/eblCgbcjBG/k8dPevU1tCWIKj/AL5r5+8aftTfCqx+H9leeM/i1pXw51aWJvtWkWF1b6ldQnkqnmRqyg9c7fXGag/Zm/aY+HL69d6MPjvr/jVNWm32M19pcgSwREyVZygAJO7PbAFZ08TUjNqpt32/AdSiuVOK2PyC/Zs/4LOfs7/s+fAbQvgz4K/Ywu7rw94Bd4op9WsbfUZ1u5CGlupHBBLSMCd2MYAA4Fe0eGf+Dm/4deO9EuvDvhvSLjSWt7MpPaw+DfNiihYgNvRWK4/hyRjn3r8WvjN8MviF4C+LPiPTvhtZ6lDo8l86W6QswDID0IHUA5x7Ve/ZF1f9pX4C/Gq1+Ifw90PV7e6mgms71o7QP58EowyMrgqwJCnB9K3w+UQxleFF4hw52lzSb5Y3aXM7Juy3dtbHRRzrCu0qlKEl17/fc/W6f/gsjpGtX40z4Zadp1280IVYYPAdsv70nO8ZQnPPToK7z4a/tKftD/ETU7TQPEur6nbpdgMmnrFHak5PUwwquPq3Wvk34QwftUfE/GtfE3xmnhm0LAiy0SwhiupgezSxouzPfb+dfdv7DPwl0rQtcivhZx4Ub1eWQtJu6lmZslmPqfWnmXD+DyjESgsSq1usYyjF+a5tWvkh18+w+IXJQpJeZ//Z\"}}, \"prediction\": [0.0406896845, 0.125281364, 0.118567884, 0.100639313, 0.12864624, 0.0898737088, 0.157521054, 0.121037535, 0.0313298739, 0.0864133239]}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_online_predictions:migration" + }, + "source": [ + "## Make online predictions\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_create:migration,new" + }, + "source": [ + "### [projects.locations.endpoints.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/create)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "O_IkMU4i_5rj" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_create:migration,new,request" + }, + "outputs": [], + "source": [ + "endpoint = {\"display_name\": \"custom_job_TF\" + TIMESTAMP}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateEndpointRequest(parent=PARENT, endpoint=endpoint).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "GD2ezZB1_5rk" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"endpoint\": {\n", + " \"displayName\": \"custom_job_TF_TF20210227173057\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sqoAv87L_5rk" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "s70F_62P_5rk" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,response" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_create:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/endpoints/6810814827095654400\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_deploymodel:migration,new" + }, + "source": [ + "### [projects.locations.endpoints.deployModel](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/deployModel)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "EN_sldlj_5rl" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_deploymodel:migration,new,request" + }, + "outputs": [], + "source": [ + "deployed_model = {\n", + " \"model\": model_id,\n", + " \"display_name\": \"custom_job_TF\" + TIMESTAMP,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": 1,\n", + " \"machine_spec\": {\"machine_type\": \"n1-standard-4\", \"accelerator_count\": 0},\n", + " },\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.DeployModelRequest(\n", + " endpoint=endpoint_id,\n", + " deployed_model=deployed_model,\n", + " traffic_split={\"0\": 100},\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0kJRbqBm_5rl" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"endpoint\": \"projects/116273516712/locations/us-central1/endpoints/6810814827095654400\",\n", + " \"deployedModel\": {\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/8844102097923211264\",\n", + " \"displayName\": \"custom_job_TF_TF20210227173057\",\n", + " \"dedicatedResources\": {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-4\"\n", + " },\n", + " \"minReplicaCount\": 1\n", + " }\n", + " },\n", + " \"trafficSplit\": {\n", + " \"0\": 100\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cZlTImIm_5rl" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_deploymodel:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint_id, deployed_model=deployed_model, traffic_split={\"0\": 100}\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bg7Dd8XM_5rm" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "CQSG7JM0_5rm" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_deploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"deployedModel\": {\n", + " \"id\": \"2064302294823862272\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deployed_model_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The unique ID for the deployed model\n", + "deployed_model_id = result.deployed_model.id\n", + "\n", + "print(deployed_model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_predict:migration,new" + }, + "source": [ + "### [projects.locations.endpoints.predict](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/predict)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "DA2OeN6e_5rn" + }, + "source": [ + "### Prepare file for online prediction" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tPDV6rxh_5ro" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_predict:migration,new,request,icn" + }, + "outputs": [], + "source": [ + "import base64\n", + "\n", + "import cv2\n", + "import tensorflow as tf\n", + "\n", + "(_, _), (x_test, y_test) = tf.keras.datasets.cifar10.load_data()\n", + "test_image, test_label = x_test[0], y_test[0]\n", + "\n", + "cv2.imwrite(\"tmp.jpg\", (test_image * 255).astype(np.uint8))\n", + "bytes = tf.io.read_file(\"tmp.jpg\")\n", + "b64str = base64.b64encode(bytes.numpy()).decode(\"utf-8\")\n", + "\n", + "instances_list = [{\"bytes_inputs\": {\"b64\": b64str}}]\n", + "\n", + "prediction_request = aip.PredictRequest(endpoint=endpoint_id)\n", + "prediction_request.instances.append(instances_list)\n", + "\n", + "print(MessageToJson(prediction_request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "9aOWkkN-_5ro" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"endpoint\": \"projects/116273516712/locations/us-central1/endpoints/6810814827095654400\",\n", + " \"instances\": [\n", + " [\n", + " {\n", + " \"bytes_inputs\": {\n", + " \"b64\": \"/9j/4AAQSkZJRgABAQAAAQABAAD/2wBDAAIBAQEBAQIBAQECAgICAgQDAgICAgUEBAMEBgUGBgYFBgYGBwkIBgcJBwYGCAsICQoKCgoKBggLDAsKDAkKCgr/2wBDAQICAgICAgUDAwUKBwYHCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgr/wAARCAAgACADASIAAhEBAxEB/8QAHwAAAQUBAQEBAQEAAAAAAAAAAAECAwQFBgcICQoL/8QAtRAAAgEDAwIEAwUFBAQAAAF9AQIDAAQRBRIhMUEGE1FhByJxFDKBkaEII0KxwRVS0fAkM2JyggkKFhcYGRolJicoKSo0NTY3ODk6Q0RFRkdISUpTVFVWV1hZWmNkZWZnaGlqc3R1dnd4eXqDhIWGh4iJipKTlJWWl5iZmqKjpKWmp6ipqrKztLW2t7i5usLDxMXGx8jJytLT1NXW19jZ2uHi4+Tl5ufo6erx8vP09fb3+Pn6/8QAHwEAAwEBAQEBAQEBAQAAAAAAAAECAwQFBgcICQoL/8QAtREAAgECBAQDBAcFBAQAAQJ3AAECAxEEBSExBhJBUQdhcRMiMoEIFEKRobHBCSMzUvAVYnLRChYkNOEl8RcYGRomJygpKjU2Nzg5OkNERUZHSElKU1RVVldYWVpjZGVmZ2hpanN0dXZ3eHl6goOEhYaHiImKkpOUlZaXmJmaoqOkpaanqKmqsrO0tba3uLm6wsPExcbHyMnK0tPU1dbX2Nna4uPk5ebn6Onq8vP09fb3+Pn6/9oADAMBAAIRAxEAPwD6E1zw/qemaZY669mkdtqsTPZTpMH85Y3KMcKeOR36444NZGj2/ibWPHaaPeHSLXRbq3jSw1O7u3V3u9zb0ZAh+QIFO4EkliCBjnwv9lfxtrviTxBbW974le/0nQ/h5ohms7m4b92bhVlkEfPIDuwJ6gyADgCuWh1fxP8As6/tGad8H5PiRrHjW6tNd1O/iXUr5Z7mx0uSZlinHODiRQCqrgGTGPmwPyqfClGlnM6Em3TSi/N3Wtnto015H6y+MK08kp14QSqScle6tFxel0+6aZ9d6/rvhXwH4407wWtq+uSXth9pa5jcwKUBIbyxzkL0Ock8nHQV2x0NtN0Gw8a6PDOunXc3liO5GZIGxwG6YBxx1x0zkV4L8Xfij4k8X/Gr4V+HdJtDpdgui3GoajJBAXlkuGvNoUEDcD5MYyuN3zEnpX0B4Q+Iunafdap8OPFCG/sL+PzLkGNgbQB1O7Jxh1JOCOvHXNfUYrh/LPqMo0oKDgvdl10117nzGD4izR5hGdWcp8zs4+umisflx8DNXi/Z/wDHviPTfiP4g+x2WieFtV03U5r9miLw2ilonTIySWijZCB6Yr2X4R/tQT/tC/s56f8AGn4C/AvxTrXiq7jksW1G78NxRlNiRxIrzO5EwiVHAePAfeoO1lIrqv2pf2Xz+1t+z3feC9E1GLSvE2paQtraa1cISXiEqu9tKVydrbMZ5Kkg8jIr234a/Bq7+EngjQPAng3wzB/ZOl6ZFa2tpp/yeWiqFB2Hq2ASeuTz15r9ixHBa+vSp1JXpxXuy6vyfpbXuz8jocUyWCVSirTb1j09V95e+E3hnwXr8dn8QPjLaSWZBguP+EcudKSW6gnSMfLHOrcQh2djCSAxY5BxkzfEDx1H4n8ZyvpEC2WnMAwighMe8hvl3gZyQCB15K5xWNq3iKbVNVk8MW91NZzxLllkt9jL2z0I/DrXCeG47T4seNL3wN4c1nULKPTY2GoX8YYNcSkfKisxwis2ASMnk9AK7f8AiHuQ47CulWlKzfM7S5W+vRfgZQ47zvA4qNako3irK8eZLpfVn//Z\"\n", + " }\n", + " }\n", + " ]\n", + " ]\n", + "}\n", + "\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "-c1HT8Bw_5ro" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_predict:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"prediction\"].predict(endpoint=endpoint_id, instances=instances_list)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "koz-wcHo_5ro" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "9zRSZ3DM_5ro" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_predict:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"predictions\": [\n", + " [\n", + " 0.0406113081,\n", + " 0.125313938,\n", + " 0.118626907,\n", + " 0.100714684,\n", + " 0.128500372,\n", + " 0.0899592042,\n", + " 0.157601,\n", + " 0.121072263,\n", + " 0.0312432405,\n", + " 0.0863570943\n", + " ]\n", + " ],\n", + " \"deployedModelId\": \"2064302294823862272\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new" + }, + "source": [ + "### [projects.locations.endpoints.undeployModel](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/undeployModel)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "x8bg9Xyj_5rp" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_undeploymodel:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint_id, deployed_model_id=deployed_model_id, traffic_split={}\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "a3_S-AC6_5rp" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gWdonkAJ_5rq" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup:migration,new" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ND2Y2TnN_5rq" + }, + "outputs": [], + "source": [ + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_custom_job = True\n", + "delete_batchjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the model using the Vertex AI fully qualified identifier for the model\n", + "try:\n", + " if delete_model:\n", + " clients[\"model\"].delete_model(name=model_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex AI fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint:\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom training using the Vertex AI fully qualified identifier for the custom training\n", + "try:\n", + " if delete_custom_job:\n", + " clients[\"job\"].delete_custom_job(name=custom_training_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex AI fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob:\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r gs://$BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "UJ2,12 unified Custom Training Prebuilt Container TF Keras.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/migration/UJ3 legacy Custom Training Custom Container (TF Keras).ipynb b/notebooks/community/migration/UJ3 legacy Custom Training Custom Container (TF Keras).ipynb new file mode 100644 index 000000000..f9ed0dd4e --- /dev/null +++ b/notebooks/community/migration/UJ3 legacy Custom Training Custom Container (TF Keras).ipynb @@ -0,0 +1,2001 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "a6b56b1c7b76" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_title:migration,automl,icn" + }, + "source": [ + "# Vertex Custom Training Custom Container TF Keras" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "JE-aKjayJjkF" + }, + "source": [ + "## Installation\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Iddo-8HmJjkH" + }, + "source": [ + "Install the Google *cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "xHvi6EFSJjkH" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-storage" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ktRTgB8DJjkI" + }, + "source": [ + "### Restart the Kernel\n", + "\n", + "Once you've installed the Vertex SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "x1g4mBdlJjkI" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "if not os.getenv(\"AUTORUN\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "hceMthWGJjkI" + }, + "source": [ + "#### Project ID\n", + "\n", + "**If you don't know your project ID**, try to get your project ID using `gcloud` command by executing the second cell below." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "d3c69d85a220" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "0UfsdLCEJjkJ" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "jT0fijejJjkK" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported forAutoML. We recommend when possible, to choose the region closest to you. \n", + "\n", + "Currently project resources must be in the `us-central1` region to use this API." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "CTBlncfrJjkK" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "J_croPBoJjkK" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "KKxStk5bJjkL" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your GCP account\n", + "\n", + "**If you are using AutoML Notebooks**, your environment is already\n", + "authenticated. Skip this step.\n", + "\n", + "*Note: If you are on an AutoML notebook and run the cell, the cell knows to skip executing the authentication steps.*\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "6aUuFIftxXiN" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your Google Cloud account. This provides access\n", + "# to your Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on AutoML, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this tutorial in a notebook locally, replace the string\n", + " # below with the path to your service account key and run this cell to\n", + " # authenticate your Google Cloud account.\n", + " else:\n", + " %env GOOGLE_APPLICATION_CREDENTIALS your_path_to_credentials.json\n", + "\n", + " # Log in to your account on Google Cloud\n", + " ! gcloud auth login" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "MDUAZaN3JjkL" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets. " + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n", + " BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "yU6VuylGJjkM" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "QVRixM4oJjkM" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Mi-Pdh6QJjkN" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aqDptKpYJjkN" + }, + "outputs": [], + "source": [ + "! gsutil ls -al gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "fR9geV9pJjkO" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "9a4f8d49a2e6" + }, + "outputs": [], + "source": [ + "import json\n", + "import time\n", + "\n", + "from googleapiclient import discovery" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "lbv411XjJjkP" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Setup up the following constants for AutoML:\n", + "\n", + "- `PARENT`: The AutoM location root path for dataset, model and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Gx8Eos8PJjkP" + }, + "outputs": [], + "source": [ + "# AutoM location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "7ce9a09da09a" + }, + "source": [ + "## Vertex API Client\n", + "\n", + "We use the Google APIs Client Library for Python to call the Vertex Training and Prediction API without manually constructing HTTP requests." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "b2d711e139ec" + }, + "outputs": [], + "source": [ + "cloudml = discovery.build(\"ml\", \"v1\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "32d560062b7d" + }, + "source": [ + "## Prepare trainer script and custom container" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "b894c0df7a31" + }, + "outputs": [], + "source": [ + "%%writefile cifar/Dockerfile\n", + "\n", + "FROM gcr.io/deeplearning-platform-release/tf2-cpu.2-1\n", + "WORKDIR /root\n", + "\n", + "WORKDIR /\n", + "\n", + "# Copies the trainer code to the docker image.\n", + "COPY trainer /trainer\n", + "\n", + "# Sets up the entry point to invoke the trainer.\n", + "ENTRYPOINT [\"python\", \"-m\", \"trainer.task\"]\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "6de4fe86cba2" + }, + "outputs": [], + "source": [ + "# Add package information\n", + "! touch cifar/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\\n", + "tag_build =\\n\\\n", + "tag_date = 0\"\n", + "! echo \"$setup_cfg\" > cifar/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\\n", + "# Requires TensorFlow Datasets\\n\\\n", + "setuptools.setup(\\n\\\n", + " install_requires=[\\n\\\n", + " 'tensorflow_datasets==1.3.0',\\n\\\n", + " ],\\n\\\n", + " packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > cifar/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\\n", + "Name: Custom Training CIFAR-10\\n\\\n", + "Version: 0.0.0\\n\\\n", + "Summary: Demonstration training script\\n\\\n", + "Home-page: www.google.com\\n\\\n", + "Author: Google\\n\\\n", + "Author-email: aferlitsch@google.com\\n\\\n", + "License: Public\\n\\\n", + "Description: Demo\\n\\\n", + "Platform: Vertex AI\"\n", + "! echo \"$pkg_info\" > cifar/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir cifar/trainer\n", + "! touch cifar/trainer/__init__.py" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "8cf1b642f91a" + }, + "outputs": [], + "source": [ + "%%writefile cifar/trainer/task.py\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "import argparse\n", + "import os\n", + "import sys\n", + "\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default='/tmp/saved_model', type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.01, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=10, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=200, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "print('DEVICES', device_lib.list_local_devices())\n", + "\n", + "# Single Machine, single compute device\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "# Single Machine, multiple compute device\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "# Multiple Machine, multiple compute device\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "# Multi-worker configuration\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "# Preparing dataset\n", + "BUFFER_SIZE = 10000\n", + "BATCH_SIZE = 64\n", + "\n", + "def make_datasets_unbatched():\n", + " # Scaling CIFAR10 data from (0, 255] to (0., 1.]\n", + " def scale(image, label):\n", + " image = tf.cast(image, tf.float32)\n", + " image /= 255.0\n", + " return image, label\n", + "\n", + " datasets, info = tfds.load(name='cifar10',\n", + " with_info=True,\n", + " as_supervised=True)\n", + " return datasets['train'].map(scale).cache().shuffle(BUFFER_SIZE).repeat()\n", + "\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_cnn_model():\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu', input_shape=(32, 32, 3)),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu'),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Flatten(),\n", + " tf.keras.layers.Dense(10, activation='softmax')\n", + " ])\n", + " model.compile(\n", + " loss=tf.keras.losses.sparse_categorical_crossentropy,\n", + " optimizer=tf.keras.optimizers.SGD(learning_rate=args.lr),\n", + " metrics=['accuracy'])\n", + " return model\n", + "\n", + "# Train the model\n", + "NUM_WORKERS = strategy.num_replicas_in_sync\n", + "# Here the batch size scales up by number of workers since\n", + "# `tf.data.Dataset.batch` expects the global batch size.\n", + "GLOBAL_BATCH_SIZE = BATCH_SIZE * NUM_WORKERS\n", + "train_dataset = make_datasets_unbatched().batch(GLOBAL_BATCH_SIZE)\n", + "\n", + "with strategy.scope():\n", + " # Creation of dataset, and model building/compiling need to be within\n", + " # `strategy.scope()`.\n", + " model = build_and_compile_cnn_model()\n", + "\n", + "model.fit(x=train_dataset, epochs=args.epochs, steps_per_epoch=args.steps)\n", + "model.save(args.model_dir)\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "a62b630e2ffd" + }, + "outputs": [], + "source": [ + "TRAIN_IMAGE = f\"gcr.io/{PROJECT_ID}/cifar_migration:v1\"\n", + "\n", + "! docker build cifar -t $TRAIN_IMAGE\n", + "! docker push $TRAIN_IMAGE" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_create_and_deploy_model:migration" + }, + "source": [ + "## Train a model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0oqIBOSnJjkW" + }, + "source": [ + "### [projects.jobs.create](https://cloud.google.com/ai-platform/training/docs/reference/rest/v1/projects.jobs/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "95e10759f288" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "2f0883073672" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_container_\" + TIMESTAMP\n", + "\n", + "TRAINING_INPUTS = {\n", + " \"scaleTier\": \"CUSTOM\",\n", + " \"masterType\": \"n1-standard-4\",\n", + " \"masterConfig\": {\"imageUri\": TRAIN_IMAGE},\n", + " \"args\": [\n", + " \"--model-dir=\" + \"gs://{}/{}\".format(BUCKET_NAME, JOB_NAME),\n", + " \"--epochs=\" + str(20),\n", + " \"--steps=\" + str(100),\n", + " ],\n", + " \"region\": REGION,\n", + "}\n", + "\n", + "body = {\"jobId\": JOB_NAME, \"trainingInput\": TRAINING_INPUTS}\n", + "\n", + "request = cloudml.projects().jobs().create(parent=PARENT)\n", + "request.body = json.loads(json.dumps(TRAINING_INPUTS, indent=2))\n", + "\n", + "print(json.dumps(json.loads(request.to_json()), indent=2))\n", + "\n", + "request = cloudml.projects().jobs().create(parent=PARENT, body=body)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dQak2NW3_5re" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"uri\": \"https://ml.googleapis.com/v1/projects/migration-ucaip-training/jobs?alt=json\",\n", + " \"method\": \"POST\",\n", + " \"body\": {\n", + " \"scaleTier\": \"CUSTOM\",\n", + " \"masterType\": \"n1-standard-4\",\n", + " \"masterConfig\": {\n", + " \"imageUri\": \"gcr.io/migration-ucaip-training/cifar_migration:v1\"\n", + " },\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210325215916/custom_container_20210325215916\",\n", + " \"--epochs=20\",\n", + " \"--steps=100\"\n", + " ],\n", + " \"region\": \"us-central1\"\n", + " },\n", + " \"headers\": {\n", + " \"accept\": \"application/json\",\n", + " \"accept-encoding\": \"gzip, deflate\",\n", + " \"user-agent\": \"(gzip)\",\n", + " \"x-goog-api-client\": \"gdcl/1.12.8 gl-python/3.7.8\"\n", + " },\n", + " \"methodId\": \"ml.projects.jobs.create\",\n", + " \"resumable\": null,\n", + " \"response_callbacks\": [],\n", + " \"_in_error_state\": false,\n", + " \"body_size\": 0,\n", + " \"resumable_uri\": null,\n", + " \"resumable_progress\": 0\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "f889d0d09ff6" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "46cbb4cb04d3" + }, + "outputs": [], + "source": [ + "response = request.execute()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "12048b2b192e" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "fe91f92693ea" + }, + "outputs": [], + "source": [ + "print(json.dumps(response, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dQak2NW3_5re" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"jobId\": \"custom_container_20210325215916\",\n", + " \"trainingInput\": {\n", + " \"scaleTier\": \"CUSTOM\",\n", + " \"masterType\": \"n1-standard-4\",\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210325215916/custom_container_20210325215916\",\n", + " \"--epochs=20\",\n", + " \"--steps=100\"\n", + " ],\n", + " \"region\": \"us-central1\",\n", + " \"masterConfig\": {\n", + " \"imageUri\": \"gcr.io/migration-ucaip-training/cifar_migration:v1\"\n", + " }\n", + " },\n", + " \"createTime\": \"2021-03-25T21:59:28Z\",\n", + " \"state\": \"QUEUED\",\n", + " \"trainingOutput\": {},\n", + " \"etag\": \"DZ8rE8+ASE4=\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "training_pipeline_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom training job\n", + "custom_training_id = f'{PARENT}/jobs/{response[\"jobId\"]}'\n", + "# The short numeric ID for the custom training job\n", + "custom_training_short_id = response[\"jobId\"]\n", + "\n", + "print(custom_training_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "791f42ba0799" + }, + "source": [ + "### [projects.jobs.get](https://cloud.google.com/ai-platform/training/docs/reference/rest/v1/projects.jobs/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "7252ca012ecb" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "d7ac6ebfd39f" + }, + "outputs": [], + "source": [ + "request = cloudml.projects().jobs().get(name=custom_training_id)\n", + "\n", + "response = request.execute()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "3aa8dc1e4927" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "18cf10ff1df2" + }, + "outputs": [], + "source": [ + "print(json.dumps(response, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dQak2NW3_5re" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"jobId\": \"custom_container_20210325215916\",\n", + " \"trainingInput\": {\n", + " \"scaleTier\": \"CUSTOM\",\n", + " \"masterType\": \"n1-standard-4\",\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210325215916/custom_container_20210325215916\",\n", + " \"--epochs=20\",\n", + " \"--steps=100\"\n", + " ],\n", + " \"region\": \"us-central1\",\n", + " \"masterConfig\": {\n", + " \"imageUri\": \"gcr.io/migration-ucaip-training/cifar_migration:v1\"\n", + " }\n", + " },\n", + " \"createTime\": \"2021-03-25T21:59:28Z\",\n", + " \"state\": \"PREPARING\",\n", + " \"trainingOutput\": {},\n", + " \"etag\": \"Nn3P/Dd/c9A=\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = cloudml.projects().jobs().get(name=custom_training_id).execute()\n", + " if response[\"state\"] != \"SUCCEEDED\":\n", + " print(\"Training job has not completed:\", response[\"state\"])\n", + " if response[\"state\"] == \"FAILED\":\n", + " break\n", + " else:\n", + " break\n", + " time.sleep(20)\n", + "\n", + "# model artifact output directory on Google Cloud Storage\n", + "model_artifact_dir = response[\"trainingInput\"][\"args\"][0].split(\"=\")[-1]\n", + "print(\"artifact location \" + model_artifact_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "1f54b9e3a2f3" + }, + "source": [ + "### Serving function for trained model (image data)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "c892fde7669d" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(model_artifact_dir)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "02bfe69618fd" + }, + "outputs": [], + "source": [ + "CONCRETE_INPUT = \"numpy_inputs\"\n", + "\n", + "\n", + "def _preprocess(bytes_input):\n", + " decoded = tf.io.decode_jpeg(bytes_input, channels=3)\n", + " decoded = tf.image.convert_image_dtype(decoded, tf.float32)\n", + " resized = tf.image.resize(decoded, size=(32, 32))\n", + " rescale = tf.cast(resized / 255.0, tf.float32)\n", + " return rescale\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def preprocess_fn(bytes_inputs):\n", + " decoded_images = tf.map_fn(\n", + " _preprocess, bytes_inputs, dtype=tf.float32, back_prop=False\n", + " )\n", + " return {\n", + " CONCRETE_INPUT: decoded_images\n", + " } # User needs to make sure the key matches model's input\n", + "\n", + "\n", + "m_call = tf.function(model.call).get_concrete_function(\n", + " [tf.TensorSpec(shape=[None, 32, 32, 3], dtype=tf.float32, name=CONCRETE_INPUT)]\n", + ")\n", + "\n", + "\n", + "@tf.function(\n", + " input_signature=[tf.TensorSpec([None], tf.string), tf.TensorSpec([None], tf.string)]\n", + ")\n", + "def serving_fn(bytes_inputs, key):\n", + " images = preprocess_fn(bytes_inputs)\n", + " prob = m_call(**images)\n", + " return {\"prediction\": prob, \"key\": key}\n", + "\n", + "\n", + "tf.saved_model.save(\n", + " model, model_artifact_dir, signatures={\"serving_default\": serving_fn}\n", + ")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "58e922e21268" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_artifact_dir)\n", + "\n", + "tensors_specs = list(loaded.signatures[\"serving_default\"].structured_input_signature)\n", + "print(\"Tensors specs:\", tensors_specs)\n", + "\n", + "input_name = [v for k, v in tensors_specs[1].items() if k != \"key\"][0].name\n", + "print(\"Bytes input tensor name:\", input_name)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "uSLjjGyy_5rd" + }, + "source": [ + "*Example output*:\n", + "```\n", + "Tensors specs: [(), {'bytes_inputs': TensorSpec(shape=(None,), dtype=tf.string, name='bytes_inputs'), 'key': TensorSpec(shape=(None,), dtype=tf.string, name='key')}]\n", + "Bytes input tensor name: bytes_inputs\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ba0fcf66032d" + }, + "source": [ + "## Make batch predictions" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "14d2c093e9f3" + }, + "source": [ + "### Prepare files for batch prediction" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "3a6efeb58681" + }, + "outputs": [], + "source": [ + "import base64\n", + "import json\n", + "\n", + "import cv2\n", + "import numpy as np\n", + "import tensorflow as tf\n", + "\n", + "(_, _), (x_test, y_test) = tf.keras.datasets.cifar10.load_data()\n", + "\n", + "test_image_1, test_label_1 = x_test[0], y_test[0]\n", + "test_image_2, test_label_2 = x_test[1], y_test[1]\n", + "\n", + "cv2.imwrite(\"tmp1.jpg\", (test_image_1 * 255).astype(np.uint8))\n", + "cv2.imwrite(\"tmp2.jpg\", (test_image_2 * 255).astype(np.uint8))\n", + "\n", + "gcs_input_uri = \"gs://\" + BUCKET_NAME + \"/\" + \"test.json\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " for img in [\"tmp1.jpg\", \"tmp2.jpg\"]:\n", + " bytes = tf.io.read_file(img)\n", + " b64str = base64.b64encode(bytes.numpy()).decode(\"utf-8\")\n", + " f.write(json.dumps({\"key\": img, input_name: {\"b64\": b64str}}) + \"\\n\")\n", + "\n", + "! gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\"key\": \"tmp1.jpg\", \"bytes_inputs\": {\"b64\": \"/9j/4AAQSkZJRgABAQAAAQABAAD/2wBDAAIBAQEBAQIBAQECAgICAgQDAgICAgUEBAMEBgUGBgYFBgYGBwkIBgcJBwYGCAsICQoKCgoKBggLDAsKDAkKCgr/2wBDAQICAgICAgUDAwUKBwYHCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgr/wAARCAAgACADASIAAhEBAxEB/8QAHwAAAQUBAQEBAQEAAAAAAAAAAAECAwQFBgcICQoL/8QAtRAAAgEDAwIEAwUFBAQAAAF9AQIDAAQRBRIhMUEGE1FhByJxFDKBkaEII0KxwRVS0fAkM2JyggkKFhcYGRolJicoKSo0NTY3ODk6Q0RFRkdISUpTVFVWV1hZWmNkZWZnaGlqc3R1dnd4eXqDhIWGh4iJipKTlJWWl5iZmqKjpKWmp6ipqrKztLW2t7i5usLDxMXGx8jJytLT1NXW19jZ2uHi4+Tl5ufo6erx8vP09fb3+Pn6/8QAHwEAAwEBAQEBAQEBAQAAAAAAAAECAwQFBgcICQoL/8QAtREAAgECBAQDBAcFBAQAAQJ3AAECAxEEBSExBhJBUQdhcRMiMoEIFEKRobHBCSMzUvAVYnLRChYkNOEl8RcYGRomJygpKjU2Nzg5OkNERUZHSElKU1RVVldYWVpjZGVmZ2hpanN0dXZ3eHl6goOEhYaHiImKkpOUlZaXmJmaoqOkpaanqKmqsrO0tba3uLm6wsPExcbHyMnK0tPU1dbX2Nna4uPk5ebn6Onq8vP09fb3+Pn6/9oADAMBAAIRAxEAPwD6E1zw/qemaZY669mkdtqsTPZTpMH85Y3KMcKeOR36444NZGj2/ibWPHaaPeHSLXRbq3jSw1O7u3V3u9zb0ZAh+QIFO4EkliCBjnwv9lfxtrviTxBbW974le/0nQ/h5ohms7m4b92bhVlkEfPIDuwJ6gyADgCuWh1fxP8As6/tGad8H5PiRrHjW6tNd1O/iXUr5Z7mx0uSZlinHODiRQCqrgGTGPmwPyqfClGlnM6Em3TSi/N3Wtnto015H6y+MK08kp14QSqScle6tFxel0+6aZ9d6/rvhXwH4407wWtq+uSXth9pa5jcwKUBIbyxzkL0Ock8nHQV2x0NtN0Gw8a6PDOunXc3liO5GZIGxwG6YBxx1x0zkV4L8Xfij4k8X/Gr4V+HdJtDpdgui3GoajJBAXlkuGvNoUEDcD5MYyuN3zEnpX0B4Q+Iunafdap8OPFCG/sL+PzLkGNgbQB1O7Jxh1JOCOvHXNfUYrh/LPqMo0oKDgvdl10117nzGD4izR5hGdWcp8zs4+umisflx8DNXi/Z/wDHviPTfiP4g+x2WieFtV03U5r9miLw2ilonTIySWijZCB6Yr2X4R/tQT/tC/s56f8AGn4C/AvxTrXiq7jksW1G78NxRlNiRxIrzO5EwiVHAePAfeoO1lIrqv2pf2Xz+1t+z3feC9E1GLSvE2paQtraa1cISXiEqu9tKVydrbMZ5Kkg8jIr234a/Bq7+EngjQPAng3wzB/ZOl6ZFa2tpp/yeWiqFB2Hq2ASeuTz15r9ixHBa+vSp1JXpxXuy6vyfpbXuz8jocUyWCVSirTb1j09V95e+E3hnwXr8dn8QPjLaSWZBguP+EcudKSW6gnSMfLHOrcQh2djCSAxY5BxkzfEDx1H4n8ZyvpEC2WnMAwighMe8hvl3gZyQCB15K5xWNq3iKbVNVk8MW91NZzxLllkt9jL2z0I/DrXCeG47T4seNL3wN4c1nULKPTY2GoX8YYNcSkfKisxwis2ASMnk9AK7f8AiHuQ47CulWlKzfM7S5W+vRfgZQ47zvA4qNako3irK8eZLpfVn//Z\"}}\n", + "{\"key\": \"tmp2.jpg\", \"bytes_inputs\": {\"b64\": \"/9j/4AAQSkZJRgABAQAAAQABAAD/2wBDAAIBAQEBAQIBAQECAgICAgQDAgICAgUEBAMEBgUGBgYFBgYGBwkIBgcJBwYGCAsICQoKCgoKBggLDAsKDAkKCgr/2wBDAQICAgICAgUDAwUKBwYHCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgr/wAARCAAgACADASIAAhEBAxEB/8QAHwAAAQUBAQEBAQEAAAAAAAAAAAECAwQFBgcICQoL/8QAtRAAAgEDAwIEAwUFBAQAAAF9AQIDAAQRBRIhMUEGE1FhByJxFDKBkaEII0KxwRVS0fAkM2JyggkKFhcYGRolJicoKSo0NTY3ODk6Q0RFRkdISUpTVFVWV1hZWmNkZWZnaGlqc3R1dnd4eXqDhIWGh4iJipKTlJWWl5iZmqKjpKWmp6ipqrKztLW2t7i5usLDxMXGx8jJytLT1NXW19jZ2uHi4+Tl5ufo6erx8vP09fb3+Pn6/8QAHwEAAwEBAQEBAQEBAQAAAAAAAAECAwQFBgcICQoL/8QAtREAAgECBAQDBAcFBAQAAQJ3AAECAxEEBSExBhJBUQdhcRMiMoEIFEKRobHBCSMzUvAVYnLRChYkNOEl8RcYGRomJygpKjU2Nzg5OkNERUZHSElKU1RVVldYWVpjZGVmZ2hpanN0dXZ3eHl6goOEhYaHiImKkpOUlZaXmJmaoqOkpaanqKmqsrO0tba3uLm6wsPExcbHyMnK0tPU1dbX2Nna4uPk5ebn6Onq8vP09fb3+Pn6/9oADAMBAAIRAxEAPwD8V5UubRQlxlSvDAtyD6dadbW91fK8lrFI6o6KzrnALHCj8cH8jX3J+1V+wR8adOjsrDxR8EPhzohsg13qKfD+zddRWBF2u0sR42AnIzjJAAzzXmnwh/Yk+D3jX4Q6h478cftgaX4Al/tR4f8AhHdf0eRruVI+IpdkbFiWLsAqgnrXZLBVFWcI6/gc0MVSlSU2eZaX+zdr954Nv/EEt7FNeWyrJHZ2moRn93tYsTuwcg7OBz19q8sa7AUEMf8AvqvoHwX+yz8Vb74gXtn4M+Euq/EbSYpV+y6vf2txptrMOAz+XIysR0xu9M4qf9pn9mf4jJoNprJ+BGgeCn0mHZfQ2OqRl793fAZUDkkAbcd8k1pUw1OUE6e/bf8AEVOs1JqT3P19/aT/AOCMf7RH7Qfx5134zeNf2z7S18Q+PkSWWDSb6406BrSMFYrWNCCAsakDbnOSSeTXg+sf8G3viHwt49ez1jxdY6zqds1veTwT+MzBdqJWnWCYb0DhXe3n2sOGMD4J2HH7IfD3xnc/EPwl4Y8R6t458M28y+EL1NRh1nS3vGXV3a1+w3S4mjCwxxpdCaFSjTNLGRImwk+A6f8AAL9oH4gaX4+tf+Ckn7Vfw4+I2k3fiW6m+HOneFNPn0WDw9piTLLbuUiYGWZsCNYp/tMtqiSbL+b7RMrqvWxVDKamZ89BOg03Q9+deupOpBRotU1CM4OMak/aSUIxkouTbUjmllc0qic60XrGNldX/dtNr/n2+aS5r3XI3ytKz+Jof+CN2r6LYHU/ibqOo2iQzFmmn8eXLfugMbDhwMcdeprg/iV+zX+zx8O9Mu9f8NaRplw9oSr6g0sl0BgdBNMzZ+i9K+svi9P+yv8ADAnRfhl4MfxNdhSDe63fzS2sJHdYpHbfjtu/KvhL9ub4tarruhy2JvJMsdjJFGFj28gKqrgKo9B6VhlvEGMzfDxm8M6N+kpRlJeT5dE/mwoZDiMO+evVb8j/2Q==\"}}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "da797816753b" + }, + "source": [ + "### [projects.jobs.create](https://cloud.google.com/ai-platform/prediction/docs/reference/rest/v1/projects.jobs/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "6119b4448a7f" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dfaf15aaf35b" + }, + "outputs": [], + "source": [ + "body = {\n", + " \"jobId\": \"custom_container_pred_\" + TIMESTAMP,\n", + " \"predictionInput\": {\n", + " \"dataFormat\": \"JSON\",\n", + " \"inputPaths\": gcs_input_uri,\n", + " \"outputPath\": \"gs://\" + f\"{BUCKET_NAME}/batch_output/\",\n", + " \"runtime_version\": \"2.1\",\n", + " \"uri\": model_artifact_dir,\n", + " \"region\": REGION,\n", + " },\n", + "}\n", + "\n", + "request = cloudml.projects().jobs().create(parent=PARENT)\n", + "request.body = json.loads(json.dumps(body, indent=2))\n", + "\n", + "print(json.dumps(json.loads(request.to_json()), indent=2))\n", + "\n", + "request = cloudml.projects().jobs().create(parent=PARENT, body=body)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"uri\": \"https://ml.googleapis.com/v1/projects/migration-ucaip-training/jobs?alt=json\",\n", + " \"method\": \"POST\",\n", + " \"body\": {\n", + " \"jobId\": \"custom_container_pred_20210325215916\",\n", + " \"predictionInput\": {\n", + " \"dataFormat\": \"JSON\",\n", + " \"inputPaths\": \"gs://migration-ucaip-trainingaip-20210325215916/test.json\",\n", + " \"outputPath\": \"gs://migration-ucaip-trainingaip-20210325215916/batch_output/\",\n", + " \"runtime_version\": \"2.1\",\n", + " \"uri\": \"gs://migration-ucaip-trainingaip-20210325215916/custom_container_20210325215916\",\n", + " \"region\": \"us-central1\"\n", + " }\n", + " },\n", + " \"headers\": {\n", + " \"accept\": \"application/json\",\n", + " \"accept-encoding\": \"gzip, deflate\",\n", + " \"user-agent\": \"(gzip)\",\n", + " \"x-goog-api-client\": \"gdcl/1.12.8 gl-python/3.7.8\"\n", + " },\n", + " \"methodId\": \"ml.projects.jobs.create\",\n", + " \"resumable\": null,\n", + " \"response_callbacks\": [],\n", + " \"_in_error_state\": false,\n", + " \"body_size\": 0,\n", + " \"resumable_uri\": null,\n", + " \"resumable_progress\": 0\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "764bbab3553c" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "0116917784fb" + }, + "outputs": [], + "source": [ + "response = request.execute()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "1e7c8645f963" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "41dd1b1b101e" + }, + "outputs": [], + "source": [ + "print(json.dumps(response, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"jobId\": \"custom_container_pred_20210325215916\",\n", + " \"predictionInput\": {\n", + " \"dataFormat\": \"JSON\",\n", + " \"inputPaths\": [\n", + " \"gs://migration-ucaip-trainingaip-20210325215916/test.json\"\n", + " ],\n", + " \"outputPath\": \"gs://migration-ucaip-trainingaip-20210325215916/batch_output/\",\n", + " \"region\": \"us-central1\",\n", + " \"runtimeVersion\": \"2.1\",\n", + " \"uri\": \"gs://migration-ucaip-trainingaip-20210325215916/custom_container_20210325215916\",\n", + " \"framework\": \"TENSORFLOW\"\n", + " },\n", + " \"createTime\": \"2021-03-25T22:15:15Z\",\n", + " \"state\": \"QUEUED\",\n", + " \"predictionOutput\": {\n", + " \"outputPath\": \"gs://migration-ucaip-trainingaip-20210325215916/batch_output/\"\n", + " },\n", + " \"etag\": \"GNq2pYok7CI=\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "8b5cc896b446" + }, + "outputs": [], + "source": [ + "# The full unique ID for the batch prediction job\n", + "batch_job_id = PARENT + \"/jobs/\" + response[\"jobId\"]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "a86f592e0e41" + }, + "source": [ + "### [projects.jobs.get](https://cloud.google.com/ai-platform/prediction/docs/reference/rest/v1/projects.jobs/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "37838e6ea3bb" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "b06849c66667" + }, + "outputs": [], + "source": [ + "request = cloudml.projects().jobs().get(name=batch_job_id)\n", + "\n", + "response = request.execute()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "3fa5feeb4c18" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "65f7bc6bb370" + }, + "outputs": [], + "source": [ + "print(json.dumps(response, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"jobId\": \"custom_container_pred_20210325215916\",\n", + " \"predictionInput\": {\n", + " \"dataFormat\": \"JSON\",\n", + " \"inputPaths\": [\n", + " \"gs://migration-ucaip-trainingaip-20210325215916/test.json\"\n", + " ],\n", + " \"outputPath\": \"gs://migration-ucaip-trainingaip-20210325215916/batch_output/\",\n", + " \"region\": \"us-central1\",\n", + " \"runtimeVersion\": \"2.1\",\n", + " \"uri\": \"gs://migration-ucaip-trainingaip-20210325215916/custom_container_20210325215916\",\n", + " \"framework\": \"TENSORFLOW\"\n", + " },\n", + " \"createTime\": \"2021-03-25T22:15:15Z\",\n", + " \"state\": \"QUEUED\",\n", + " \"predictionOutput\": {\n", + " \"outputPath\": \"gs://migration-ucaip-trainingaip-20210325215916/batch_output/\"\n", + " },\n", + " \"etag\": \"Sxnlx4MEtTo=\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = request = cloudml.projects().jobs().get(name=batch_job_id).execute()\n", + " if response[\"state\"] != \"SUCCEEDED\":\n", + " print(\"The job has not completed:\", response[\"state\"])\n", + " if response[\"state\"] == \"FAILED\":\n", + " break\n", + " else:\n", + " folder = response[\"predictionInput\"][\"outputPath\"][:-1]\n", + " ! gsutil ls $folder/prediction*\n", + "\n", + " ! gsutil cat $folder/prediction*\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210325215916/batch_output/prediction.errors_stats-00000-of-00001\n", + "gs://migration-ucaip-trainingaip-20210325215916/batch_output/prediction.results-00000-of-00001\n", + "{\"prediction\": [0.04647013917565346, 0.06366760283708572, 0.1313525140285492, 0.11146997660398483, 0.1568831354379654, 0.09669718891382217, 0.18583332002162933, 0.10817062109708786, 0.03371051326394081, 0.06574499607086182], \"key\": \"tmp1.jpg\"}\n", + "{\"prediction\": [0.04657613858580589, 0.06360984593629837, 0.13138002157211304, 0.11128606647253036, 0.15718042850494385, 0.096551313996315, 0.1853194385766983, 0.10867659002542496, 0.03375411406159401, 0.06566616892814636], \"key\": \"tmp2.jpg\"}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "e5a08f8923cf" + }, + "source": [ + "## Make online predictions" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "41fe6501624f" + }, + "source": [ + "### Deploy the model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "8870a668ac3b" + }, + "source": [ + "### [projects.models.create](https://cloud.google.com/ai-platform/prediction/docs/reference/rest/v1/projects.models/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "d58086b4c67c" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ce70e710bfa5" + }, + "outputs": [], + "source": [ + "request = cloudml.projects().models().create(parent=PARENT)\n", + "request.body = json.loads(\n", + " json.dumps({\"name\": \"custom_container_\" + TIMESTAMP}, indent=2)\n", + ")\n", + "\n", + "print(json.dumps(json.loads(request.to_json()), indent=2))\n", + "\n", + "request = (\n", + " cloudml.projects()\n", + " .models()\n", + " .create(parent=PARENT, body={\"name\": \"custom_container_\" + TIMESTAMP})\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dQak2NW3_5re" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"uri\": \"https://ml.googleapis.com/v1/projects/migration-ucaip-training/models?alt=json\",\n", + " \"method\": \"POST\",\n", + " \"body\": {\n", + " \"name\": \"custom_container_20210325215916\"\n", + " },\n", + " \"headers\": {\n", + " \"accept\": \"application/json\",\n", + " \"accept-encoding\": \"gzip, deflate\",\n", + " \"user-agent\": \"(gzip)\",\n", + " \"x-goog-api-client\": \"gdcl/1.12.8 gl-python/3.7.8\"\n", + " },\n", + " \"methodId\": \"ml.projects.models.create\",\n", + " \"resumable\": null,\n", + " \"response_callbacks\": [],\n", + " \"_in_error_state\": false,\n", + " \"body_size\": 0,\n", + " \"resumable_uri\": null,\n", + " \"resumable_progress\": 0\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "02db2acdfc6a" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "7697208c5ad5" + }, + "outputs": [], + "source": [ + "response = request.execute()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "505977fbd87d" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "a537b1fab166" + }, + "outputs": [], + "source": [ + "print(json.dumps(response, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dQak2NW3_5re" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/migration-ucaip-training/models/custom_container_20210325215916\",\n", + " \"regions\": [\n", + " \"us-central1\"\n", + " ],\n", + " \"etag\": \"gBP35vWqHPE=\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "zOvD5qZK_5re" + }, + "outputs": [], + "source": [ + "# The full unique ID for the training pipeline\n", + "model_id = response[\"name\"]\n", + "# The short numeric ID for the training pipeline\n", + "model_short_name = model_id.split(\"/\")[-1]\n", + "\n", + "print(model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0e2424beb653" + }, + "source": [ + "### [projects.models.versions.create](https://cloud.google.com/ai-platform/prediction/docs/reference/rest/v1/projects.models.versions/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "72d35ff235e6" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "3314888fe5ae" + }, + "outputs": [], + "source": [ + "version = {\n", + " \"name\": \"custom_container_\" + TIMESTAMP,\n", + " \"deploymentUri\": model_artifact_dir,\n", + " \"runtimeVersion\": \"2.1\",\n", + " \"framework\": \"TENSORFLOW\",\n", + " \"pythonVersion\": \"3.7\",\n", + " \"machineType\": \"mls1-c1-m2\",\n", + "}\n", + "\n", + "request = cloudml.projects().models().versions().create(parent=response[\"name\"])\n", + "request.body = json.loads(json.dumps(version, indent=2))\n", + "\n", + "print(json.dumps(json.loads(request.to_json()), indent=2))\n", + "\n", + "request = (\n", + " cloudml.projects().models().versions().create(parent=response[\"name\"], body=version)\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dQak2NW3_5re" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"uri\": \"https://ml.googleapis.com/v1/projects/migration-ucaip-training/models/custom_container_20210325215916/versions?alt=json\",\n", + " \"method\": \"POST\",\n", + " \"body\": {\n", + " \"name\": \"custom_container_20210325215916\",\n", + " \"deploymentUri\": \"gs://migration-ucaip-trainingaip-20210325215916/custom_container_20210325215916\",\n", + " \"runtimeVersion\": \"2.1\",\n", + " \"framework\": \"TENSORFLOW\",\n", + " \"pythonVersion\": \"3.7\",\n", + " \"machineType\": \"mls1-c1-m2\"\n", + " },\n", + " \"headers\": {\n", + " \"accept\": \"application/json\",\n", + " \"accept-encoding\": \"gzip, deflate\",\n", + " \"user-agent\": \"(gzip)\",\n", + " \"x-goog-api-client\": \"gdcl/1.12.8 gl-python/3.7.8\"\n", + " },\n", + " \"methodId\": \"ml.projects.models.versions.create\",\n", + " \"resumable\": null,\n", + " \"response_callbacks\": [],\n", + " \"_in_error_state\": false,\n", + " \"body_size\": 0,\n", + " \"resumable_uri\": null,\n", + " \"resumable_progress\": 0\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "49a7c1b56d12" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "d05c3fc1cf03" + }, + "outputs": [], + "source": [ + "response = request.execute()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "848fcda5efa8" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "207ba27ba372" + }, + "outputs": [], + "source": [ + "print(json.dumps(response, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dQak2NW3_5re" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/migration-ucaip-training/operations/create_custom_container_20210325215916_custom_container_20210325215916-1616710881327\",\n", + " \"metadata\": {\n", + " \"@type\": \"type.googleapis.com/google.cloud.ml.v1.OperationMetadata\",\n", + " \"createTime\": \"2021-03-25T22:21:21Z\",\n", + " \"operationType\": \"CREATE_VERSION\",\n", + " \"modelName\": \"projects/migration-ucaip-training/models/custom_container_20210325215916\",\n", + " \"version\": {\n", + " \"name\": \"projects/migration-ucaip-training/models/custom_container_20210325215916/versions/custom_container_20210325215916\",\n", + " \"deploymentUri\": \"gs://migration-ucaip-trainingaip-20210325215916/custom_container_20210325215916\",\n", + " \"createTime\": \"2021-03-25T22:21:21Z\",\n", + " \"runtimeVersion\": \"2.1\",\n", + " \"etag\": \"d2zy+bRwFOw=\",\n", + " \"framework\": \"TENSORFLOW\",\n", + " \"machineType\": \"mls1-c1-m2\",\n", + " \"pythonVersion\": \"3.7\"\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "9c2afb3ebcca" + }, + "outputs": [], + "source": [ + "# The full unique ID for the model version\n", + "model_version_name = response[\"metadata\"][\"version\"][\"name\"]\n", + "\n", + "print(model_version_name)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "b02e301a722a" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = (\n", + " cloudml.projects().models().versions().get(name=model_version_name).execute()\n", + " )\n", + " if response[\"state\"] == \"READY\":\n", + " print(\"Model version created.\")\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sqbcNfhKyIs-" + }, + "source": [ + "### Prepare input for online prediction" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "2769f68b7ee2" + }, + "outputs": [], + "source": [ + "import base64\n", + "import json\n", + "\n", + "import cv2\n", + "import tensorflow as tf\n", + "\n", + "(_, _), (x_test, y_test) = tf.keras.datasets.cifar10.load_data()\n", + "\n", + "test_image_1, test_label_1 = x_test[0], y_test[0]\n", + "test_image_2, test_label_2 = x_test[1], y_test[1]\n", + "\n", + "cv2.imwrite(\"tmp1.jpg\", (test_image_1 * 255).astype(np.uint8))\n", + "cv2.imwrite(\"tmp2.jpg\", (test_image_2 * 255).astype(np.uint8))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "f8da5c2e6645" + }, + "source": [ + "### [projects.predict](https://cloud.google.com/ai-platform/prediction/docs/reference/rest/v1/projects/predict)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "54b7e2145fdf" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "296e506782ec" + }, + "outputs": [], + "source": [ + "instances_list = []\n", + "for img in [\"tmp1.jpg\", \"tmp2.jpg\"]:\n", + " bytes = tf.io.read_file(img)\n", + " b64str = base64.b64encode(bytes.numpy()).decode(\"utf-8\")\n", + " instances_list.append({\"key\": img, input_name: {\"b64\": b64str}})\n", + "\n", + "request = cloudml.projects().predict(name=model_version_name)\n", + "request.body = json.loads(json.dumps({\"instances\": instances_list}, indent=2))\n", + "\n", + "print(json.dumps(json.loads(request.to_json()), indent=2))\n", + "\n", + "request = cloudml.projects().predict(\n", + " name=model_version_name, body={\"instances\": instances_list}\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"uri\": \"https://ml.googleapis.com/v1/projects/migration-ucaip-training/models/custom_container_20210325215916/versions/custom_container_20210325215916:predict?alt=json\",\n", + " \"method\": \"POST\",\n", + " \"body\": {\n", + " \"instances\": [\n", + " {\n", + " \"key\": \"tmp1.jpg\",\n", + " \"bytes_inputs\": {\n", + " \"b64\": \"/9j/4AAQSkZJRgABAQAAAQABAAD/2wBDAAIBAQEBAQIBAQECAgICAgQDAgICAgUEBAMEBgUGBgYFBgYGBwkIBgcJBwYGCAsICQoKCgoKBggLDAsKDAkKCgr/2wBDAQICAgICAgUDAwUKBwYHCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgr/wAARCAAgACADASIAAhEBAxEB/8QAHwAAAQUBAQEBAQEAAAAAAAAAAAECAwQFBgcICQoL/8QAtRAAAgEDAwIEAwUFBAQAAAF9AQIDAAQRBRIhMUEGE1FhByJxFDKBkaEII0KxwRVS0fAkM2JyggkKFhcYGRolJicoKSo0NTY3ODk6Q0RFRkdISUpTVFVWV1hZWmNkZWZnaGlqc3R1dnd4eXqDhIWGh4iJipKTlJWWl5iZmqKjpKWmp6ipqrKztLW2t7i5usLDxMXGx8jJytLT1NXW19jZ2uHi4+Tl5ufo6erx8vP09fb3+Pn6/8QAHwEAAwEBAQEBAQEBAQAAAAAAAAECAwQFBgcICQoL/8QAtREAAgECBAQDBAcFBAQAAQJ3AAECAxEEBSExBhJBUQdhcRMiMoEIFEKRobHBCSMzUvAVYnLRChYkNOEl8RcYGRomJygpKjU2Nzg5OkNERUZHSElKU1RVVldYWVpjZGVmZ2hpanN0dXZ3eHl6goOEhYaHiImKkpOUlZaXmJmaoqOkpaanqKmqsrO0tba3uLm6wsPExcbHyMnK0tPU1dbX2Nna4uPk5ebn6Onq8vP09fb3+Pn6/9oADAMBAAIRAxEAPwD6E1zw/qemaZY669mkdtqsTPZTpMH85Y3KMcKeOR36444NZGj2/ibWPHaaPeHSLXRbq3jSw1O7u3V3u9zb0ZAh+QIFO4EkliCBjnwv9lfxtrviTxBbW974le/0nQ/h5ohms7m4b92bhVlkEfPIDuwJ6gyADgCuWh1fxP8As6/tGad8H5PiRrHjW6tNd1O/iXUr5Z7mx0uSZlinHODiRQCqrgGTGPmwPyqfClGlnM6Em3TSi/N3Wtnto015H6y+MK08kp14QSqScle6tFxel0+6aZ9d6/rvhXwH4407wWtq+uSXth9pa5jcwKUBIbyxzkL0Ock8nHQV2x0NtN0Gw8a6PDOunXc3liO5GZIGxwG6YBxx1x0zkV4L8Xfij4k8X/Gr4V+HdJtDpdgui3GoajJBAXlkuGvNoUEDcD5MYyuN3zEnpX0B4Q+Iunafdap8OPFCG/sL+PzLkGNgbQB1O7Jxh1JOCOvHXNfUYrh/LPqMo0oKDgvdl10117nzGD4izR5hGdWcp8zs4+umisflx8DNXi/Z/wDHviPTfiP4g+x2WieFtV03U5r9miLw2ilonTIySWijZCB6Yr2X4R/tQT/tC/s56f8AGn4C/AvxTrXiq7jksW1G78NxRlNiRxIrzO5EwiVHAePAfeoO1lIrqv2pf2Xz+1t+z3feC9E1GLSvE2paQtraa1cISXiEqu9tKVydrbMZ5Kkg8jIr234a/Bq7+EngjQPAng3wzB/ZOl6ZFa2tpp/yeWiqFB2Hq2ASeuTz15r9ixHBa+vSp1JXpxXuy6vyfpbXuz8jocUyWCVSirTb1j09V95e+E3hnwXr8dn8QPjLaSWZBguP+EcudKSW6gnSMfLHOrcQh2djCSAxY5BxkzfEDx1H4n8ZyvpEC2WnMAwighMe8hvl3gZyQCB15K5xWNq3iKbVNVk8MW91NZzxLllkt9jL2z0I/DrXCeG47T4seNL3wN4c1nULKPTY2GoX8YYNcSkfKisxwis2ASMnk9AK7f8AiHuQ47CulWlKzfM7S5W+vRfgZQ47zvA4qNako3irK8eZLpfVn//Z\"\n", + " }\n", + " },\n", + " {\n", + " \"key\": \"tmp2.jpg\",\n", + " \"bytes_inputs\": {\n", + " \"b64\": \"/9j/4AAQSkZJRgABAQAAAQABAAD/2wBDAAIBAQEBAQIBAQECAgICAgQDAgICAgUEBAMEBgUGBgYFBgYGBwkIBgcJBwYGCAsICQoKCgoKBggLDAsKDAkKCgr/2wBDAQICAgICAgUDAwUKBwYHCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgr/wAARCAAgACADASIAAhEBAxEB/8QAHwAAAQUBAQEBAQEAAAAAAAAAAAECAwQFBgcICQoL/8QAtRAAAgEDAwIEAwUFBAQAAAF9AQIDAAQRBRIhMUEGE1FhByJxFDKBkaEII0KxwRVS0fAkM2JyggkKFhcYGRolJicoKSo0NTY3ODk6Q0RFRkdISUpTVFVWV1hZWmNkZWZnaGlqc3R1dnd4eXqDhIWGh4iJipKTlJWWl5iZmqKjpKWmp6ipqrKztLW2t7i5usLDxMXGx8jJytLT1NXW19jZ2uHi4+Tl5ufo6erx8vP09fb3+Pn6/8QAHwEAAwEBAQEBAQEBAQAAAAAAAAECAwQFBgcICQoL/8QAtREAAgECBAQDBAcFBAQAAQJ3AAECAxEEBSExBhJBUQdhcRMiMoEIFEKRobHBCSMzUvAVYnLRChYkNOEl8RcYGRomJygpKjU2Nzg5OkNERUZHSElKU1RVVldYWVpjZGVmZ2hpanN0dXZ3eHl6goOEhYaHiImKkpOUlZaXmJmaoqOkpaanqKmqsrO0tba3uLm6wsPExcbHyMnK0tPU1dbX2Nna4uPk5ebn6Onq8vP09fb3+Pn6/9oADAMBAAIRAxEAPwD8V5UubRQlxlSvDAtyD6dadbW91fK8lrFI6o6KzrnALHCj8cH8jX3J+1V+wR8adOjsrDxR8EPhzohsg13qKfD+zddRWBF2u0sR42AnIzjJAAzzXmnwh/Yk+D3jX4Q6h478cftgaX4Al/tR4f8AhHdf0eRruVI+IpdkbFiWLsAqgnrXZLBVFWcI6/gc0MVSlSU2eZaX+zdr954Nv/EEt7FNeWyrJHZ2moRn93tYsTuwcg7OBz19q8sa7AUEMf8AvqvoHwX+yz8Vb74gXtn4M+Euq/EbSYpV+y6vf2txptrMOAz+XIysR0xu9M4qf9pn9mf4jJoNprJ+BGgeCn0mHZfQ2OqRl793fAZUDkkAbcd8k1pUw1OUE6e/bf8AEVOs1JqT3P19/aT/AOCMf7RH7Qfx5134zeNf2z7S18Q+PkSWWDSb6406BrSMFYrWNCCAsakDbnOSSeTXg+sf8G3viHwt49ez1jxdY6zqds1veTwT+MzBdqJWnWCYb0DhXe3n2sOGMD4J2HH7IfD3xnc/EPwl4Y8R6t458M28y+EL1NRh1nS3vGXV3a1+w3S4mjCwxxpdCaFSjTNLGRImwk+A6f8AAL9oH4gaX4+tf+Ckn7Vfw4+I2k3fiW6m+HOneFNPn0WDw9piTLLbuUiYGWZsCNYp/tMtqiSbL+b7RMrqvWxVDKamZ89BOg03Q9+deupOpBRotU1CM4OMak/aSUIxkouTbUjmllc0qic60XrGNldX/dtNr/n2+aS5r3XI3ytKz+Jof+CN2r6LYHU/ibqOo2iQzFmmn8eXLfugMbDhwMcdeprg/iV+zX+zx8O9Mu9f8NaRplw9oSr6g0sl0BgdBNMzZ+i9K+svi9P+yv8ADAnRfhl4MfxNdhSDe63fzS2sJHdYpHbfjtu/KvhL9ub4tarruhy2JvJMsdjJFGFj28gKqrgKo9B6VhlvEGMzfDxm8M6N+kpRlJeT5dE/mwoZDiMO+evVb8j/2Q==\"\n", + " }\n", + " }\n", + " ]\n", + " },\n", + " \"headers\": {\n", + " \"accept\": \"application/json\",\n", + " \"accept-encoding\": \"gzip, deflate\",\n", + " \"user-agent\": \"(gzip)\",\n", + " \"x-goog-api-client\": \"gdcl/1.12.8 gl-python/3.7.8\"\n", + " },\n", + " \"methodId\": \"ml.projects.predict\",\n", + " \"resumable\": null,\n", + " \"response_callbacks\": [],\n", + " \"_in_error_state\": false,\n", + " \"body_size\": 0,\n", + " \"resumable_uri\": null,\n", + " \"resumable_progress\": 0\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "1d692130a3c5" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "368499863a68" + }, + "outputs": [], + "source": [ + "response = request.execute()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "34843864fb82" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "055d8811155c" + }, + "outputs": [], + "source": [ + "print(json.dumps(response, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"predictions\": [\n", + " {\n", + " \"key\": \"tmp1.jpg\",\n", + " \"prediction\": [\n", + " 0.04647013917565346,\n", + " 0.06366760283708572,\n", + " 0.1313525289297104,\n", + " 0.11146996915340424,\n", + " 0.15688316524028778,\n", + " 0.09669718891382217,\n", + " 0.18583332002162933,\n", + " 0.10817062109708786,\n", + " 0.03371051698923111,\n", + " 0.06574499607086182\n", + " ]\n", + " },\n", + " {\n", + " \"key\": \"tmp2.jpg\",\n", + " \"prediction\": [\n", + " 0.04657613858580589,\n", + " 0.06360984593629837,\n", + " 0.13138002157211304,\n", + " 0.11128604412078857,\n", + " 0.15718042850494385,\n", + " 0.09655129164457321,\n", + " 0.1853194385766983,\n", + " 0.10867657512426376,\n", + " 0.03375410661101341,\n", + " 0.06566616147756577\n", + " ]\n", + " }\n", + " ]\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "f55142162f55" + }, + "source": [ + "### [projects.models.versions.delete](https://cloud.google.com/ai-platform/prediction/docs/reference/rest/v1/projects.models.versions/delete)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "94a516980d63" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "4888d98b38d6" + }, + "outputs": [], + "source": [ + "request = cloudml.projects().models().versions().delete(name=model_version_name)\n", + "\n", + "response = request.execute()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "response:migration" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,request" + }, + "outputs": [], + "source": [ + "print(json.dumps(response, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/migration-ucaip-training/operations/delete_custom_container_20210325215916_custom_container_20210325215916-1616710943615\",\n", + " \"metadata\": {\n", + " \"@type\": \"type.googleapis.com/google.cloud.ml.v1.OperationMetadata\",\n", + " \"createTime\": \"2021-03-25T22:22:23Z\",\n", + " \"operationType\": \"DELETE_VERSION\",\n", + " \"modelName\": \"projects/migration-ucaip-training/models/custom_container_20210325215916\",\n", + " \"version\": {\n", + " \"name\": \"projects/migration-ucaip-training/models/custom_container_20210325215916/versions/custom_container_20210325215916\",\n", + " \"deploymentUri\": \"gs://migration-ucaip-trainingaip-20210325215916/custom_container_20210325215916\",\n", + " \"createTime\": \"2021-03-25T22:21:21Z\",\n", + " \"runtimeVersion\": \"2.1\",\n", + " \"state\": \"READY\",\n", + " \"etag\": \"kfUhdXr8GRg=\",\n", + " \"framework\": \"TENSORFLOW\",\n", + " \"machineType\": \"mls1-c1-m2\",\n", + " \"pythonVersion\": \"3.7\"\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bQ-VVaSxJjkd" + }, + "source": [ + "# Cleanup" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "rbFGEzYPyIs_" + }, + "outputs": [], + "source": [ + "delete_model = True\n", + "delete_bucket = True\n", + "\n", + "\n", + "# Delete the model using the Vertex AI fully qualified identifier for the model\n", + "try:\n", + " if delete_model:\n", + " cloudml.projects().models().delete(name=model_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r gs://$BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "UJ3 legacy Custom Training Custom Container (TF Keras).ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/migration/UJ3 unified Custom Training Custom Container (TF Keras).ipynb b/notebooks/community/migration/UJ3 unified Custom Training Custom Container (TF Keras).ipynb new file mode 100644 index 000000000..b618f2c3c --- /dev/null +++ b/notebooks/community/migration/UJ3 unified Custom Training Custom Container (TF Keras).ipynb @@ -0,0 +1,2333 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title:migration,new" + }, + "source": [ + "# Vertex SDK: Train & deploy a TensorFlow model with custom container (aka pre-built containers)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest (preview) version of Vertex SDK.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "KEMtN2uGdx7-" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-aiplatform --user" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the Google *cloud-storage* library as well.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dHmqkk5jdx7-" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-storage" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the Kernel\n", + "\n", + "Once you've installed the Vertex SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "067J3q5Wdx7_" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "if not os.getenv(\"AUTORUN\") and False:\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU run-time\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your GCP project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a GCP project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebooks.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex AI. We recommend when possible, to choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You cannot use a Multi-Regional Storage bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see [Region support for Vertex AI services](https://cloud.google.com/vertex-ai/docs/general/locations)\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "mH8isSmUdx8B" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "rP3ppCOxdx8C" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your GCP account\n", + "\n", + "**If you are using Google Cloud Notebooks**, your environment is already\n", + "authenticated. Skip this step.\n", + "\n", + "*Note: If you are on an Vertex notebook and run the cell, the cell knows to skip executing the authentication steps.*\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "_0rTTHRRdx8C" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your Google Cloud account. This provides access\n", + "# to your Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Vertex, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this tutorial in a notebook locally, replace the string\n", + " # below with the path to your service account key and run this cell to\n", + " # authenticate your Google Cloud account.\n", + " else:\n", + " %env GOOGLE_APPLICATION_CREDENTIALS your_path_to_credentials.json\n", + "\n", + " # Log in to your account on Google Cloud\n", + " ! gcloud auth login" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n", + " BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "iFVSyRUndx8E" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "WSUveW5Ydx8E" + }, + "outputs": [], + "source": [ + "! gsutil ls -al gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip" + }, + "source": [ + "#### Import Vertex SDK\n", + "\n", + "Import the Vertex SDK into our Python environment.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "_WA8wIGjdx8F" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex AI constants\n", + "\n", + "Setup up the following constants for Vertex AI:\n", + "\n", + "- `API_ENDPOINT`: The Vertex AI API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `API_PREDICT_ENDPOINT`: The Vertex AI API service endpoint for prediction.\n", + "- `PARENT`: The Vertex AI location root path for dataset, model and endpoint resources.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "rBI-1o0Rdx8F" + }, + "outputs": [], + "source": [ + "# API Endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex AI location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients" + }, + "source": [ + "## Clients\n", + "\n", + "The Vertex SDK works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the server (Vertex).\n", + "\n", + "You will use several clients in this tutorial, so set them all up upfront.\n", + "\n", + "- Dataset Service for managed datasets.\n", + "- Model Service for managed models.\n", + "- Pipeline Service for training.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving. *Note*: Prediction has a different service endpoint.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "7OswurIxdx8G" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "clients[\"job\"] = create_job_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bBZ62eSJdx8I" + }, + "source": [ + "## Prepare a trainer script" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "8eNZwmHIdx8I" + }, + "source": [ + "### Package assembly\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "vSF5zUqddx8J" + }, + "outputs": [], + "source": [ + "! rm -rf cifar\n", + "! mkdir cifar\n", + "! touch cifar/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\\n", + "tag_build =\\n\\\n", + "tag_date = 0\"\n", + "! echo \"$setup_cfg\" > cifar/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\\n", + "# Requires TensorFlow Datasets\\n\\\n", + "setuptools.setup(\\n\\\n", + " install_requires=[\\n\\\n", + " 'tensorflow_datasets==1.3.0',\\n\\\n", + " ],\\n\\\n", + " packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > cifar/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\\n", + "Name: Custom Training CIFAR-10\\n\\\n", + "Version: 0.0.0\\n\\\n", + "Summary: Demonstration training script\\n\\\n", + "Home-page: www.google.com\\n\\\n", + "Author: Google\\n\\\n", + "Author-email: aferlitsch@google.com\\n\\\n", + "License: Public\\n\\\n", + "Description: Demo\\n\\\n", + "Platform: Vertex AI\"\n", + "! echo \"$pkg_info\" > cifar/PKG-INFO\n", + "\n", + "! mkdir cifar/trainer\n", + "! touch cifar/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "-zNmWOxvdx8J" + }, + "source": [ + "### Write the docker file contents\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "KohUv929dx8J" + }, + "outputs": [], + "source": [ + "%%writefile cifar/Dockerfile\n", + "\n", + "FROM gcr.io/deeplearning-platform-release/tf2-cpu.2-1\n", + "WORKDIR /root\n", + "\n", + "WORKDIR /\n", + "\n", + "# Copies the trainer code to the docker image.\n", + "COPY trainer /trainer\n", + "\n", + "# Sets up the entry point to invoke the trainer.\n", + "ENTRYPOINT [\"python\", \"-m\", \"trainer.task\"]\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "TALbjjJ6dx8K" + }, + "source": [ + "### Task.py contents" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "x1ZpANzodx8L" + }, + "outputs": [], + "source": [ + "%%writefile cifar/trainer/task.py\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "from tensorflow.python.client import device_lib\n", + "import argparse\n", + "import os\n", + "import sys\n", + "\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default='/tmp/saved_model', type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.01, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=10, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=200, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "print('DEVICES', device_lib.list_local_devices())\n", + "\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "BUFFER_SIZE = 10000\n", + "BATCH_SIZE = 64\n", + "\n", + "def make_datasets_unbatched():\n", + " def scale(image, label):\n", + " image = tf.cast(image, tf.float32)\n", + " image /= 255.0\n", + " return image, label\n", + "\n", + " datasets, info = tfds.load(name='cifar10',\n", + " with_info=True,\n", + " as_supervised=True)\n", + " return datasets['train'].map(scale).cache().shuffle(BUFFER_SIZE).repeat()\n", + "\n", + "def build_and_compile_cnn_model():\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu', input_shape=(32, 32, 3)),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Conv2D(32, 3, activation='relu'),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Flatten(),\n", + " tf.keras.layers.Dense(10, activation='softmax')\n", + " ])\n", + " model.compile(\n", + " loss=tf.keras.losses.sparse_categorical_crossentropy,\n", + " optimizer=tf.keras.optimizers.SGD(learning_rate=args.lr),\n", + " metrics=['accuracy'])\n", + " return model\n", + "\n", + "NUM_WORKERS = strategy.num_replicas_in_sync\n", + "GLOBAL_BATCH_SIZE = BATCH_SIZE * NUM_WORKERS\n", + "train_dataset = make_datasets_unbatched().batch(GLOBAL_BATCH_SIZE)\n", + "\n", + "with strategy.scope():\n", + " model = build_and_compile_cnn_model()\n", + "\n", + "model.fit(x=train_dataset, epochs=args.epochs, steps_per_epoch=args.steps)\n", + "model.save(args.model_dir)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ZE6B22qpdx8L" + }, + "source": [ + "### Build the container locally" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "TYzH5IFkdx8L" + }, + "outputs": [], + "source": [ + "TRAIN_IMAGE = f\"gcr.io/{PROJECT_ID}/cifar_migration:v1\"\n", + "\n", + "! docker build cifar -t $TRAIN_IMAGE" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "OYJPD8uLdx8M" + }, + "source": [ + "### Register your custom container" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "fCxMAQKZdx8M" + }, + "outputs": [], + "source": [ + "! docker push $TRAIN_IMAGE" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_create_and_deploy_model:migration" + }, + "source": [ + "## Train a model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0oqIBOSnJjkW" + }, + "source": [ + "### [projects.locations.customJobs.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.trainingPipelines/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "WWsBQks0dx8N" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Pefe5OQOdx8O" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_container_\" + TIMESTAMP\n", + "\n", + "WORKER_POOL_SPEC = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": {\"machine_type\": \"n1-standard-4\", \"accelerator_count\": 0},\n", + " \"container_spec\": {\n", + " \"image_uri\": TRAIN_IMAGE,\n", + " \"args\": [\n", + " \"--model-dir=\" + \"gs://\" + BUCKET_NAME + \"/\" + JOB_NAME,\n", + " \"--epochs=\" + str(20),\n", + " \"--steps=\" + str(100),\n", + " ],\n", + " },\n", + " }\n", + "]\n", + "\n", + "CUSTOM_JOB = {\n", + " \"display_name\": JOB_NAME,\n", + " \"job_spec\": {\"worker_pool_specs\": WORKER_POOL_SPEC},\n", + "}\n", + "\n", + "training_job = aip.CustomJob(**CUSTOM_JOB)\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateCustomJobRequest(parent=PARENT, custom_job=training_job).__dict__[\n", + " \"_pb\"\n", + " ]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"customJob\": {\n", + " \"displayName\": \"custom_container_20210226022223\",\n", + " \"jobSpec\": {\n", + " \"workerPoolSpecs\": [\n", + " {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-4\"\n", + " },\n", + " \"replicaCount\": \"1\",\n", + " \"containerSpec\": {\n", + " \"imageUri\": \"gcr.io/migration-ucaip-training/cifar_migration:v1\",\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210226022223/custom_container_20210226022223\",\n", + " \"--epochs=20\",\n", + " \"--steps=100\"\n", + " ]\n", + " }\n", + " }\n", + " ]\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Mul--swidx8P" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "7CulTGVSdx8P" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].create_custom_job(parent=PARENT, custom_job=training_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gy3LcQ3ydx8P" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "uRql9mxvdx8P" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sQ4EYFoqdx8P" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/customJobs/957560278583607296\",\n", + " \"displayName\": \"custom_container_20210226022223\",\n", + " \"jobSpec\": {\n", + " \"workerPoolSpecs\": [\n", + " {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-4\"\n", + " },\n", + " \"replicaCount\": \"1\",\n", + " \"diskSpec\": {\n", + " \"bootDiskType\": \"pd-ssd\",\n", + " \"bootDiskSizeGb\": 100\n", + " },\n", + " \"containerSpec\": {\n", + " \"imageUri\": \"gcr.io/migration-ucaip-training/cifar_migration:v1\",\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210226022223/custom_container_20210226022223\",\n", + " \"--epochs=20\",\n", + " \"--steps=100\"\n", + " ]\n", + " }\n", + " }\n", + " ]\n", + " },\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"createTime\": \"2021-02-26T02:27:53.406955Z\",\n", + " \"updateTime\": \"2021-02-26T02:27:53.406955Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "training_pipeline_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom training job\n", + "custom_training_id = request.name\n", + "# The short numeric ID for the custom training job\n", + "custom_training_short_id = custom_training_id.split(\"/\")[-1]\n", + "\n", + "print(custom_training_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "L2VLjDD1dx8Q" + }, + "source": [ + "### [projects.locations.customJobs.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.trainingPipelines/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "N78Cr-wKdx8Q" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "OiX3rQsrdx8Q" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].get_custom_job(name=custom_training_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "CFvGgOKrdx8R" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "_CtVywL8dx8R" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "XEdTFuXpdx8R" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/customJobs/957560278583607296\",\n", + " \"displayName\": \"custom_container_20210226022223\",\n", + " \"jobSpec\": {\n", + " \"workerPoolSpecs\": [\n", + " {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-4\"\n", + " },\n", + " \"replicaCount\": \"1\",\n", + " \"diskSpec\": {\n", + " \"bootDiskType\": \"pd-ssd\",\n", + " \"bootDiskSizeGb\": 100\n", + " },\n", + " \"containerSpec\": {\n", + " \"imageUri\": \"gcr.io/migration-ucaip-training/cifar_migration:v1\",\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210226022223/custom_container_20210226022223\",\n", + " \"--epochs=20\",\n", + " \"--steps=100\"\n", + " ]\n", + " }\n", + " }\n", + " ]\n", + " },\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"createTime\": \"2021-02-26T02:27:53.406955Z\",\n", + " \"updateTime\": \"2021-02-26T02:27:53.406955Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = clients[\"job\"].get_custom_job(name=custom_training_id)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " break\n", + " else:\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "# model artifact output directory on Google Cloud Storage\n", + "model_artifact_dir = (\n", + " response.job_spec.worker_pool_specs[0].container_spec.args[0].split(\"=\")[-1]\n", + ")\n", + "print(\"artifact location \" + model_artifact_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bAWHUeCadx8S" + }, + "source": [ + "## Deploy the model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sRjgCGMjdx8T" + }, + "source": [ + "### Load the saved model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "nFThw0Nwdx8T" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "model = tf.keras.models.load_model(model_artifact_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Ch-ZOEqVdx8T" + }, + "source": [ + "### Serving function for image data" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bJoK5szQdx8U" + }, + "outputs": [], + "source": [ + "CONCRETE_INPUT = \"numpy_inputs\"\n", + "\n", + "\n", + "def _preprocess(bytes_input):\n", + " decoded = tf.io.decode_jpeg(bytes_input, channels=3)\n", + " decoded = tf.image.convert_image_dtype(decoded, tf.float32)\n", + " resized = tf.image.resize(decoded, size=(32, 32))\n", + " rescale = tf.cast(resized / 255.0, tf.float32)\n", + " return rescale\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def preprocess_fn(bytes_inputs):\n", + " decoded_images = tf.map_fn(\n", + " _preprocess, bytes_inputs, dtype=tf.float32, back_prop=False\n", + " )\n", + " return {\n", + " CONCRETE_INPUT: decoded_images\n", + " } # User needs to make sure the key matches model's input\n", + "\n", + "\n", + "m_call = tf.function(model.call).get_concrete_function(\n", + " [tf.TensorSpec(shape=[None, 32, 32, 3], dtype=tf.float32, name=CONCRETE_INPUT)]\n", + ")\n", + "\n", + "\n", + "@tf.function(input_signature=[tf.TensorSpec([None], tf.string)])\n", + "def serving_fn(bytes_inputs):\n", + " images = preprocess_fn(bytes_inputs)\n", + " prob = m_call(**images)\n", + " return prob\n", + "\n", + "\n", + "tf.saved_model.save(\n", + " model,\n", + " model_artifact_dir,\n", + " signatures={\n", + " \"serving_default\": serving_fn,\n", + " },\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "eg_QFIr0dx8U" + }, + "source": [ + "### Get the serving function signature" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "H0F-bxlNdx8U" + }, + "outputs": [], + "source": [ + "loaded = tf.saved_model.load(model_artifact_dir)\n", + "\n", + "input_name = list(\n", + " loaded.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "\n", + "print(\"Serving function input:\", input_name)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "Serving function input: bytes_inputs\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "COwVZtxhJjkW" + }, + "source": [ + "### [projects.locations.models.upload](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.models/upload)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "z7J-ijeydx8V" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aawZj13Idx8W" + }, + "outputs": [], + "source": [ + "container_spec = {\n", + " \"image_uri\": \"gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-1:latest\",\n", + " \"env\": [{\"name\": \"exmple_env_name\", \"value\": \"example_env_value\"}],\n", + " \"ports\": [{\"container_port\": 8080}],\n", + "}\n", + "\n", + "model = {\n", + " \"display_name\": \"custom_container_TF\" + TIMESTAMP,\n", + " \"metadata_schema_uri\": \"\",\n", + " \"artifact_uri\": model_artifact_dir,\n", + " \"container_spec\": container_spec,\n", + "}\n", + "\n", + "print(MessageToJson(aip.UploadModelRequest(parent=PARENT, model=model).__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "M8HZOo3Xdx8W" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"model\": {\n", + " \"displayName\": \"custom_container_TF20210226022223\",\n", + " \"containerSpec\": {\n", + " \"imageUri\": \"gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-1:latest\",\n", + " \"env\": [\n", + " {\n", + " \"name\": \"example_env_name\",\n", + " \"value\": \"example_env_value\"\n", + " }\n", + " ],\n", + " \"ports\": [\n", + " {\n", + " \"containerPort\": 8080\n", + " }\n", + " ]\n", + " },\n", + " \"artifactUri\": \"gs://migration-ucaip-trainingaip-20210226022223/custom_container_20210226022223\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cLIJ-zE8dx8W" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "R57zQ1Ctdx8W" + }, + "outputs": [], + "source": [ + "request = clients[\"model\"].upload_model(parent=PARENT, model=model)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "QpkjnQY4dx8X" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "RLpp9Wy1dx8X" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "GiVosZjqdx9F" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/394223297069318144\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "fmYZPWbKdx9G" + }, + "outputs": [], + "source": [ + "model_id = result.model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_predictions:migration" + }, + "source": [ + "## Make batch predictions\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_prediction_file:migration,new" + }, + "source": [ + "### Make a batch prediction file\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_items:automl,icn,csv" + }, + "outputs": [], + "source": [ + "import cv2\n", + "import numpy as np\n", + "from tensorflow.keras.datasets import cifar10\n", + "\n", + "(_, _), (x_test, y_test) = cifar10.load_data()\n", + "x_test = (x_test / 255.0).astype(np.float32)\n", + "\n", + "print(x_test.shape, y_test.shape)\n", + "\n", + "test_image_1, test_label_1 = x_test[0], y_test[0]\n", + "test_image_2, test_label_2 = x_test[1], y_test[1]\n", + "\n", + "cv2.imwrite(\"tmp1.jpg\", (test_image_1 * 255).astype(np.uint8))\n", + "cv2.imwrite(\"tmp2.jpg\", (test_image_2 * 255).astype(np.uint8))\n", + "\n", + "! gsutil cp tmp1.jpg gs://$BUCKET_NAME/tmp1.jpg\n", + "! gsutil cp tmp2.jpg gs://$BUCKET_NAME/tmp2.jpg\n", + "\n", + "test_item_1 = \"gs://\" + BUCKET_NAME + \"/\" + \"tmp1.jpg\"\n", + "test_item_2 = \"gs://\" + BUCKET_NAME + \"/\" + \"tmp2.jpg\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:automl,image" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "Let's now make a batch input file, which you will store in your local Cloud Storage bucket. The batch input file can be either CSV or JSONL. You will use JSONL in this tutorial. For JSONL file, you make one dictionary entry per line for each data item (instance). The dictionary contains the key/value pairs:\n", + "\n", + "- `content`: The Cloud Storage path to the image.\n", + "- `mime_type`: The content type. In our example, it is an `jpeg` file.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "HgMBCvNOdx9H" + }, + "outputs": [], + "source": [ + "import base64\n", + "import json\n", + "\n", + "gcs_input_uri = \"gs://\" + BUCKET_NAME + \"/\" + \"test.jsonl\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " bytes = tf.io.read_file(test_item_1)\n", + " b64str = base64.b64encode(bytes.numpy()).decode(\"utf-8\")\n", + " data = {input_name: {\"b64\": b64str}}\n", + " f.write(json.dumps(data) + \"\\n\")\n", + "\n", + " bytes = tf.io.read_file(test_item_2)\n", + " b64str = base64.b64encode(bytes.numpy()).decode(\"utf-8\")\n", + " data = {input_name: {\"b64\": b64str}}\n", + " f.write(json.dumps(data) + \"\\n\")\n", + "\n", + "! gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "P6GeKGjVdx9I" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\"bytes_inputs\": {\"b64\": \"/9j/4AAQSkZJRgABAQAAAQABAAD/2wBDAAIBAQEBAQIBAQECAgICAgQDAgICAgUEBAMEBgUGBgYFBgYGBwkIBgcJBwYGCAsICQoKCgoKBggLDAsKDAkKCgr/2wBDAQICAgICAgUDAwUKBwYHCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgr/wAARCAAgACADASIAAhEBAxEB/8QAHwAAAQUBAQEBAQEAAAAAAAAAAAECAwQFBgcICQoL/8QAtRAAAgEDAwIEAwUFBAQAAAF9AQIDAAQRBRIhMUEGE1FhByJxFDKBkaEII0KxwRVS0fAkM2JyggkKFhcYGRolJicoKSo0NTY3ODk6Q0RFRkdISUpTVFVWV1hZWmNkZWZnaGlqc3R1dnd4eXqDhIWGh4iJipKTlJWWl5iZmqKjpKWmp6ipqrKztLW2t7i5usLDxMXGx8jJytLT1NXW19jZ2uHi4+Tl5ufo6erx8vP09fb3+Pn6/8QAHwEAAwEBAQEBAQEBAQAAAAAAAAECAwQFBgcICQoL/8QAtREAAgECBAQDBAcFBAQAAQJ3AAECAxEEBSExBhJBUQdhcRMiMoEIFEKRobHBCSMzUvAVYnLRChYkNOEl8RcYGRomJygpKjU2Nzg5OkNERUZHSElKU1RVVldYWVpjZGVmZ2hpanN0dXZ3eHl6goOEhYaHiImKkpOUlZaXmJmaoqOkpaanqKmqsrO0tba3uLm6wsPExcbHyMnK0tPU1dbX2Nna4uPk5ebn6Onq8vP09fb3+Pn6/9oADAMBAAIRAxEAPwD570PxBpmp6nfaEl48lzpUqpewPCU8lpEDqMsOeD26Z55Fa+s3HhnR/Aj6xZjV7rWrW4ke/wBMtLRGRLTaux1cuPnLlhtIAAUEE5490/ao8E6F4b8P3NxZeGksNW1z4h62Iby2t1/eC3ZoozJxwSiKQOhEZJ5JrqZtI8MftFfs56j8YI/hvo/gq1u9C0ywlbTbFoLa+1SOFWlgPGRmNiQzNkiPOflyf1WHFdark0K8UlUbkvJWel1vqmn5n5MuD6MM7qUJzbpxUXazvJSWtmuzTR8iaBoXirx54H1Hxo10mhx2V/8AZltpEE7ByAV8w8YLdRjAHAz1NcSNcXUtev8AwVrE0DajaQ+YZLY4jnXPJXrkjPPTPXGDXvXwi+F3hvwh8Ffip4i1a7GqX7a1b6fp0c84SKO3Wz3FiCdpHnSHDZ2/KAOtfP8A4v8Ah1qOoWul/Efwu4sL+wk8u2IkUi7JRhtwM5RgBkHpz0xXy+F4gzNY6Mqs3NTfvR6a6adj6bGcPZX/AGfKFKEYcqupemurufqP8c9Il/aA8BeHNS+HHh/7Ze634p0rUtMhsFWUJNdsFlR8HAAWWRXBPrmvGvi5+y/B+z1+0ZqHwW+PXx08LaL4VtJI75dOtPEksgfe8krskKIDCZWdCUkyU2MRuVga5X9lr9qAfsk/tCWPjTW9Ol1XwzpurtdXei27gBJTEyJcxBsDcu/OOAwBHBwa8S+JXxltPi3431/x34y8TT/2tqmpy3V1d6h8/mOzFiN46LkgDpgcdOK/HcPxo/qMalONqkn70ei816307I/Xa/C0XjXTrO8EtJdfR/cUfiz4m8aaBJefD/4NXcd4CJ7f/hI7bVXitZ4HkPzSQMvMxRUUTAEqFGCM4EPw/wDAsnhjwZEmrzte6ipKmWeYSbAV+bYTjAJBPTgNjNbOk+HYdL0qPxPcWsN5BK2FaO43q3fHUH8eld34kku/hP4LsvHPiPRtPvZNSkU6fYSFStvED8zsqjLsq5IBwOB1Jri/4iFn2BxSq0Yxulyq8eZLp1f4ms+BMkx2FlRquVm7u0uVvrbRH//Z\"}}\n", + "{\"bytes_inputs\": {\"b64\": \"/9j/4AAQSkZJRgABAQAAAQABAAD/2wBDAAIBAQEBAQIBAQECAgICAgQDAgICAgUEBAMEBgUGBgYFBgYGBwkIBgcJBwYGCAsICQoKCgoKBggLDAsKDAkKCgr/2wBDAQICAgICAgUDAwUKBwYHCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgr/wAARCAAgACADASIAAhEBAxEB/8QAHwAAAQUBAQEBAQEAAAAAAAAAAAECAwQFBgcICQoL/8QAtRAAAgEDAwIEAwUFBAQAAAF9AQIDAAQRBRIhMUEGE1FhByJxFDKBkaEII0KxwRVS0fAkM2JyggkKFhcYGRolJicoKSo0NTY3ODk6Q0RFRkdISUpTVFVWV1hZWmNkZWZnaGlqc3R1dnd4eXqDhIWGh4iJipKTlJWWl5iZmqKjpKWmp6ipqrKztLW2t7i5usLDxMXGx8jJytLT1NXW19jZ2uHi4+Tl5ufo6erx8vP09fb3+Pn6/8QAHwEAAwEBAQEBAQEBAQAAAAAAAAECAwQFBgcICQoL/8QAtREAAgECBAQDBAcFBAQAAQJ3AAECAxEEBSExBhJBUQdhcRMiMoEIFEKRobHBCSMzUvAVYnLRChYkNOEl8RcYGRomJygpKjU2Nzg5OkNERUZHSElKU1RVVldYWVpjZGVmZ2hpanN0dXZ3eHl6goOEhYaHiImKkpOUlZaXmJmaoqOkpaanqKmqsrO0tba3uLm6wsPExcbHyMnK0tPU1dbX2Nna4uPk5ebn6Onq8vP09fb3+Pn6/9oADAMBAAIRAxEAPwD9qIntrti9vhg3KkLwR69Kbc3FrYskd1LGjOjsqNjJCjLH8Mj8xXw3+yr+3v8ABbUZL2/8L/G/4ja2L0raac/xAvEbTmndtyLFKOd5AwcZwCSccV6X8Xv22/jD4K+L2n+BPA/7H+qeP4v7LSb/AISLQNYjW0ieTmWLfIoUBQiksxA6VxwxtN0VOWn4nTPC1Y1XBHpuqftI6BZ+MrDw/FZSw2dyzRyXl3p8g/eblCgbcjBG/k8dPevU1tCWIKj/AL5r5+8aftTfCqx+H9leeM/i1pXw51aWJvtWkWF1b6ldQnkqnmRqyg9c7fXGag/Zm/aY+HL69d6MPjvr/jVNWm32M19pcgSwREyVZygAJO7PbAFZ08TUjNqpt32/AdSiuVOK2PyC/Zs/4LOfs7/s+fAbQvgz4K/Ywu7rw94Bd4op9WsbfUZ1u5CGlupHBBLSMCd2MYAA4Fe0eGf+Dm/4deO9EuvDvhvSLjSWt7MpPaw+DfNiihYgNvRWK4/hyRjn3r8WvjN8MviF4C+LPiPTvhtZ6lDo8l86W6QswDID0IHUA5x7Ve/ZF1f9pX4C/Gq1+Ifw90PV7e6mgms71o7QP58EowyMrgqwJCnB9K3w+UQxleFF4hw52lzSb5Y3aXM7Juy3dtbHRRzrCu0qlKEl17/fc/W6f/gsjpGtX40z4Zadp1280IVYYPAdsv70nO8ZQnPPToK7z4a/tKftD/ETU7TQPEur6nbpdgMmnrFHak5PUwwquPq3Wvk34QwftUfE/GtfE3xmnhm0LAiy0SwhiupgezSxouzPfb+dfdv7DPwl0rQtcivhZx4Ub1eWQtJu6lmZslmPqfWnmXD+DyjESgsSq1usYyjF+a5tWvkh18+w+IXJQpJeZ//Z\"}}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_create:migration,new" + }, + "source": [ + "### [projects.locations.batchPredictionJobs.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.batchPredictionJobs/create)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "request:migration" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_create:migration,new,request,icn" + }, + "outputs": [], + "source": [ + "batch_prediction_job = {\n", + " \"display_name\": \"custom_container_TF\" + TIMESTAMP,\n", + " \"model\": model_id,\n", + " \"input_config\": {\n", + " \"instances_format\": \"jsonl\",\n", + " \"gcs_source\": {\"uris\": [gcs_input_uri]},\n", + " },\n", + " \"model_parameters\": ParseDict(\n", + " {\"confidenceThreshold\": 0.5, \"maxPredictions\": 2}, Value()\n", + " ),\n", + " \"output_config\": {\n", + " \"predictions_format\": \"jsonl\",\n", + " \"gcs_destination\": {\n", + " \"output_uri_prefix\": \"gs://\" + f\"{BUCKET_NAME}/batch_output/\"\n", + " },\n", + " },\n", + " \"dedicated_resources\": {\n", + " \"machine_spec\": {\"machine_type\": \"n1-standard-2\", \"accelerator_type\": 0},\n", + " \"starting_replica_count\": 1,\n", + " \"max_replica_count\": 1,\n", + " },\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateBatchPredictionJobRequest(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "uTo1w7CQdx9J" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"batchPredictionJob\": {\n", + " \"displayName\": \"custom_container_TF20210226022223\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/394223297069318144\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210226022223/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"modelParameters\": {\n", + " \"confidenceThreshold\": 0.5,\n", + " \"maxPredictions\": 2.0\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210226022223/batch_output/\"\n", + " }\n", + " },\n", + " \"dedicatedResources\": {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-2\"\n", + " },\n", + " \"startingReplicaCount\": 1,\n", + " \"maxReplicaCount\": 1\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "call:migration" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].create_batch_prediction_job(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "response:migration" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,request" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_create:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/batchPredictionJobs/2465140253845880832\",\n", + " \"displayName\": \"custom_container_TF20210226022223\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/394223297069318144\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210226022223/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"modelParameters\": {\n", + " \"maxPredictions\": 2.0,\n", + " \"confidenceThreshold\": 0.5\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210226022223/batch_output/\"\n", + " }\n", + " },\n", + " \"dedicatedResources\": {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-2\"\n", + " },\n", + " \"startingReplicaCount\": 1,\n", + " \"maxReplicaCount\": 1\n", + " },\n", + " \"manualBatchTuningParameters\": {},\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"createTime\": \"2021-02-26T09:39:46.357554Z\",\n", + " \"updateTime\": \"2021-02-26T09:39:46.357554Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batch_job_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The fully qualified ID for the batch job\n", + "batch_job_id = request.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split(\"/\")[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_get:migration,new" + }, + "source": [ + "### [projects.locations.batchPredictionJobs.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.batchPredictionJobs/get)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "hdp39iMPdx9M" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_get:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].get_batch_prediction_job(name=batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "FyPU67HYdx9N" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "SIX1qajtdx9N" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_get:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/batchPredictionJobs/2465140253845880832\",\n", + " \"displayName\": \"custom_container_TF20210226022223\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/394223297069318144\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210226022223/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"modelParameters\": {\n", + " \"confidenceThreshold\": 0.5,\n", + " \"maxPredictions\": 2.0\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210226022223/batch_output/\"\n", + " }\n", + " },\n", + " \"dedicatedResources\": {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-2\"\n", + " },\n", + " \"startingReplicaCount\": 1,\n", + " \"maxReplicaCount\": 1\n", + " },\n", + " \"manualBatchTuningParameters\": {},\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"createTime\": \"2021-02-26T09:39:46.357554Z\",\n", + " \"updateTime\": \"2021-02-26T09:39:46.357554Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "def get_latest_predictions(gcs_out_dir):\n", + " \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n", + " folders = !gsutil ls $gcs_out_dir\n", + " latest = \"\"\n", + " for folder in folders:\n", + " subfolder = folder.split(\"/\")[-2]\n", + " if subfolder.startswith(\"prediction-\"):\n", + " if subfolder > latest:\n", + " latest = folder[:-1]\n", + " return latest\n", + "\n", + "\n", + "while True:\n", + " response = clients[\"job\"].get_batch_prediction_job(name=batch_job_id)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", response.state)\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " folder = get_latest_predictions(\n", + " response.output_config.gcs_destination.output_uri_prefix\n", + " )\n", + " ! gsutil ls $folder/prediction*\n", + "\n", + " ! gsutil cat $folder/prediction*\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_get:migration,new,wait,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210226022223/batch_output/prediction-custom_container_TF20210226022223-2021_02_26T01_39_46_305Z/prediction.errors_stats-00000-of-00001\n", + "gs://migration-ucaip-trainingaip-20210226022223/batch_output/prediction-custom_container_TF20210226022223-2021_02_26T01_39_46_305Z/prediction.results-00000-of-00001\n", + "{\"instance\": {\"bytes_inputs\": {\"b64\": \"/9j/4AAQSkZJRgABAQAAAQABAAD/2wBDAAIBAQEBAQIBAQECAgICAgQDAgICAgUEBAMEBgUGBgYFBgYGBwkIBgcJBwYGCAsICQoKCgoKBggLDAsKDAkKCgr/2wBDAQICAgICAgUDAwUKBwYHCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgr/wAARCAAgACADASIAAhEBAxEB/8QAHwAAAQUBAQEBAQEAAAAAAAAAAAECAwQFBgcICQoL/8QAtRAAAgEDAwIEAwUFBAQAAAF9AQIDAAQRBRIhMUEGE1FhByJxFDKBkaEII0KxwRVS0fAkM2JyggkKFhcYGRolJicoKSo0NTY3ODk6Q0RFRkdISUpTVFVWV1hZWmNkZWZnaGlqc3R1dnd4eXqDhIWGh4iJipKTlJWWl5iZmqKjpKWmp6ipqrKztLW2t7i5usLDxMXGx8jJytLT1NXW19jZ2uHi4+Tl5ufo6erx8vP09fb3+Pn6/8QAHwEAAwEBAQEBAQEBAQAAAAAAAAECAwQFBgcICQoL/8QAtREAAgECBAQDBAcFBAQAAQJ3AAECAxEEBSExBhJBUQdhcRMiMoEIFEKRobHBCSMzUvAVYnLRChYkNOEl8RcYGRomJygpKjU2Nzg5OkNERUZHSElKU1RVVldYWVpjZGVmZ2hpanN0dXZ3eHl6goOEhYaHiImKkpOUlZaXmJmaoqOkpaanqKmqsrO0tba3uLm6wsPExcbHyMnK0tPU1dbX2Nna4uPk5ebn6Onq8vP09fb3+Pn6/9oADAMBAAIRAxEAPwD570PxBpmp6nfaEl48lzpUqpewPCU8lpEDqMsOeD26Z55Fa+s3HhnR/Aj6xZjV7rWrW4ke/wBMtLRGRLTaux1cuPnLlhtIAAUEE5490/ao8E6F4b8P3NxZeGksNW1z4h62Iby2t1/eC3ZoozJxwSiKQOhEZJ5JrqZtI8MftFfs56j8YI/hvo/gq1u9C0ywlbTbFoLa+1SOFWlgPGRmNiQzNkiPOflyf1WHFdark0K8UlUbkvJWel1vqmn5n5MuD6MM7qUJzbpxUXazvJSWtmuzTR8iaBoXirx54H1Hxo10mhx2V/8AZltpEE7ByAV8w8YLdRjAHAz1NcSNcXUtev8AwVrE0DajaQ+YZLY4jnXPJXrkjPPTPXGDXvXwi+F3hvwh8Ffip4i1a7GqX7a1b6fp0c84SKO3Wz3FiCdpHnSHDZ2/KAOtfP8A4v8Ah1qOoWul/Efwu4sL+wk8u2IkUi7JRhtwM5RgBkHpz0xXy+F4gzNY6Mqs3NTfvR6a6adj6bGcPZX/AGfKFKEYcqupemurufqP8c9Il/aA8BeHNS+HHh/7Ze634p0rUtMhsFWUJNdsFlR8HAAWWRXBPrmvGvi5+y/B+z1+0ZqHwW+PXx08LaL4VtJI75dOtPEksgfe8krskKIDCZWdCUkyU2MRuVga5X9lr9qAfsk/tCWPjTW9Ol1XwzpurtdXei27gBJTEyJcxBsDcu/OOAwBHBwa8S+JXxltPi3431/x34y8TT/2tqmpy3V1d6h8/mOzFiN46LkgDpgcdOK/HcPxo/qMalONqkn70ei816307I/Xa/C0XjXTrO8EtJdfR/cUfiz4m8aaBJefD/4NXcd4CJ7f/hI7bVXitZ4HkPzSQMvMxRUUTAEqFGCM4EPw/wDAsnhjwZEmrzte6ipKmWeYSbAV+bYTjAJBPTgNjNbOk+HYdL0qPxPcWsN5BK2FaO43q3fHUH8eld34kku/hP4LsvHPiPRtPvZNSkU6fYSFStvED8zsqjLsq5IBwOB1Jri/4iFn2BxSq0Yxulyq8eZLp1f4ms+BMkx2FlRquVm7u0uVvrbRH//Z\"}}, \"prediction\": [0.0441863872, 0.0965465382, 0.131534964, 0.111121729, 0.133242682, 0.0896093622, 0.160808876, 0.116257414, 0.0309254956, 0.0857665]}\n", + "{\"instance\": {\"bytes_inputs\": {\"b64\": \"/9j/4AAQSkZJRgABAQAAAQABAAD/2wBDAAIBAQEBAQIBAQECAgICAgQDAgICAgUEBAMEBgUGBgYFBgYGBwkIBgcJBwYGCAsICQoKCgoKBggLDAsKDAkKCgr/2wBDAQICAgICAgUDAwUKBwYHCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgr/wAARCAAgACADASIAAhEBAxEB/8QAHwAAAQUBAQEBAQEAAAAAAAAAAAECAwQFBgcICQoL/8QAtRAAAgEDAwIEAwUFBAQAAAF9AQIDAAQRBRIhMUEGE1FhByJxFDKBkaEII0KxwRVS0fAkM2JyggkKFhcYGRolJicoKSo0NTY3ODk6Q0RFRkdISUpTVFVWV1hZWmNkZWZnaGlqc3R1dnd4eXqDhIWGh4iJipKTlJWWl5iZmqKjpKWmp6ipqrKztLW2t7i5usLDxMXGx8jJytLT1NXW19jZ2uHi4+Tl5ufo6erx8vP09fb3+Pn6/8QAHwEAAwEBAQEBAQEBAQAAAAAAAAECAwQFBgcICQoL/8QAtREAAgECBAQDBAcFBAQAAQJ3AAECAxEEBSExBhJBUQdhcRMiMoEIFEKRobHBCSMzUvAVYnLRChYkNOEl8RcYGRomJygpKjU2Nzg5OkNERUZHSElKU1RVVldYWVpjZGVmZ2hpanN0dXZ3eHl6goOEhYaHiImKkpOUlZaXmJmaoqOkpaanqKmqsrO0tba3uLm6wsPExcbHyMnK0tPU1dbX2Nna4uPk5ebn6Onq8vP09fb3+Pn6/9oADAMBAAIRAxEAPwD9qIntrti9vhg3KkLwR69Kbc3FrYskd1LGjOjsqNjJCjLH8Mj8xXw3+yr+3v8ABbUZL2/8L/G/4ja2L0raac/xAvEbTmndtyLFKOd5AwcZwCSccV6X8Xv22/jD4K+L2n+BPA/7H+qeP4v7LSb/AISLQNYjW0ieTmWLfIoUBQiksxA6VxwxtN0VOWn4nTPC1Y1XBHpuqftI6BZ+MrDw/FZSw2dyzRyXl3p8g/eblCgbcjBG/k8dPevU1tCWIKj/AL5r5+8aftTfCqx+H9leeM/i1pXw51aWJvtWkWF1b6ldQnkqnmRqyg9c7fXGag/Zm/aY+HL69d6MPjvr/jVNWm32M19pcgSwREyVZygAJO7PbAFZ08TUjNqpt32/AdSiuVOK2PyC/Zs/4LOfs7/s+fAbQvgz4K/Ywu7rw94Bd4op9WsbfUZ1u5CGlupHBBLSMCd2MYAA4Fe0eGf+Dm/4deO9EuvDvhvSLjSWt7MpPaw+DfNiihYgNvRWK4/hyRjn3r8WvjN8MviF4C+LPiPTvhtZ6lDo8l86W6QswDID0IHUA5x7Ve/ZF1f9pX4C/Gq1+Ifw90PV7e6mgms71o7QP58EowyMrgqwJCnB9K3w+UQxleFF4hw52lzSb5Y3aXM7Juy3dtbHRRzrCu0qlKEl17/fc/W6f/gsjpGtX40z4Zadp1280IVYYPAdsv70nO8ZQnPPToK7z4a/tKftD/ETU7TQPEur6nbpdgMmnrFHak5PUwwquPq3Wvk34QwftUfE/GtfE3xmnhm0LAiy0SwhiupgezSxouzPfb+dfdv7DPwl0rQtcivhZx4Ub1eWQtJu6lmZslmPqfWnmXD+DyjESgsSq1usYyjF+a5tWvkh18+w+IXJQpJeZ//Z\"}}, \"prediction\": [0.0441891, 0.0966139063, 0.131601468, 0.111363865, 0.133115292, 0.0897044092, 0.160883322, 0.115729697, 0.0310073923, 0.0857914686]}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_online_predictions:migration" + }, + "source": [ + "## Make online predictions\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_create:migration,new" + }, + "source": [ + "### [projects.locations.endpoints.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/create)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "3Md5r_ytdx9P" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_create:migration,new,request" + }, + "outputs": [], + "source": [ + "endpoint = {\"display_name\": \"custom_container_TF\" + TIMESTAMP}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateEndpointRequest(parent=PARENT, endpoint=endpoint).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "__Sqn83udx9P" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"endpoint\": {\n", + " \"displayName\": \"custom_container_TF20210226022223\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "FIjM1WQVdx9P" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "6wQWT-6Zdx9P" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,response" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_create:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/endpoints/2977125644296519680\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_deploymodel:migration,new" + }, + "source": [ + "### [projects.locations.endpoints.deployModel](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/deployModel)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "UYdBoNpWdx9Q" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_deploymodel:migration,new,request" + }, + "outputs": [], + "source": [ + "deployed_model = {\n", + " \"model\": model_id,\n", + " \"display_name\": \"custom_container_TF\" + TIMESTAMP,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": 1,\n", + " \"machine_spec\": {\"machine_type\": \"n1-standard-4\", \"accelerator_count\": 0},\n", + " },\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.DeployModelRequest(\n", + " endpoint=endpoint_id,\n", + " deployed_model=deployed_model,\n", + " traffic_split={\"0\": 100},\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wR4HXvHHdx9R" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"endpoint\": \"projects/116273516712/locations/us-central1/endpoints/2977125644296519680\",\n", + " \"deployedModel\": {\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/394223297069318144\",\n", + " \"displayName\": \"custom_container_TF20210226022223\",\n", + " \"dedicatedResources\": {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-4\"\n", + " },\n", + " \"minReplicaCount\": 1\n", + " }\n", + " },\n", + " \"trafficSplit\": {\n", + " \"0\": 100\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "HQmSPoszdx9R" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_deploymodel:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint_id, deployed_model=deployed_model, traffic_split={\"0\": 100}\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "z99BIGGsdx9R" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "MPuA3Eoidx9R" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_deploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"deployedModel\": {\n", + " \"id\": \"1297564458264035328\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deployed_model_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The unique ID for the deployed model\n", + "deployed_model_id = result.deployed_model.id\n", + "\n", + "print(deployed_model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_predict:migration,new" + }, + "source": [ + "### [projects.locations.endpoints.predict](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/predict)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "03LnDUvjdx9S" + }, + "source": [ + "### Prepare file for online prediction" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Rje-QRuYdx9T" + }, + "outputs": [], + "source": [ + "import base64\n", + "\n", + "import cv2\n", + "\n", + "test_image = x_test[0]\n", + "test_label = y_test[0]\n", + "\n", + "print(test_image.shape)\n", + "\n", + "cv2.imwrite(\"tmp.jpg\", (test_image * 255).astype(np.uint8))\n", + "bytes = tf.io.read_file(\"tmp.jpg\")\n", + "b64str = base64.b64encode(bytes.numpy()).decode(\"utf-8\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "r1Tk5DVkdx9T" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_predict:migration,new,request,icn" + }, + "outputs": [], + "source": [ + "instances_list = [{\"bytes_inputs\": {\"b64\": b64str}}]\n", + "\n", + "prediction_request = aip.PredictRequest(endpoint=endpoint_id)\n", + "prediction_request.instances.append(instances_list)\n", + "\n", + "print(MessageToJson(prediction_request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "FGY9u-IZdx9U" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"endpoint\": \"projects/116273516712/locations/us-central1/endpoints/2977125644296519680\",\n", + " \"instances\": [\n", + " [\n", + " {\n", + " \"bytes_inputs\": {\n", + " \"b64\": \"/9j/4AAQSkZJRgABAQAAAQABAAD/2wBDAAIBAQEBAQIBAQECAgICAgQDAgICAgUEBAMEBgUGBgYFBgYGBwkIBgcJBwYGCAsICQoKCgoKBggLDAsKDAkKCgr/2wBDAQICAgICAgUDAwUKBwYHCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgoKCgr/wAARCAAgACADASIAAhEBAxEB/8QAHwAAAQUBAQEBAQEAAAAAAAAAAAECAwQFBgcICQoL/8QAtRAAAgEDAwIEAwUFBAQAAAF9AQIDAAQRBRIhMUEGE1FhByJxFDKBkaEII0KxwRVS0fAkM2JyggkKFhcYGRolJicoKSo0NTY3ODk6Q0RFRkdISUpTVFVWV1hZWmNkZWZnaGlqc3R1dnd4eXqDhIWGh4iJipKTlJWWl5iZmqKjpKWmp6ipqrKztLW2t7i5usLDxMXGx8jJytLT1NXW19jZ2uHi4+Tl5ufo6erx8vP09fb3+Pn6/8QAHwEAAwEBAQEBAQEBAQAAAAAAAAECAwQFBgcICQoL/8QAtREAAgECBAQDBAcFBAQAAQJ3AAECAxEEBSExBhJBUQdhcRMiMoEIFEKRobHBCSMzUvAVYnLRChYkNOEl8RcYGRomJygpKjU2Nzg5OkNERUZHSElKU1RVVldYWVpjZGVmZ2hpanN0dXZ3eHl6goOEhYaHiImKkpOUlZaXmJmaoqOkpaanqKmqsrO0tba3uLm6wsPExcbHyMnK0tPU1dbX2Nna4uPk5ebn6Onq8vP09fb3+Pn6/9oADAMBAAIRAxEAPwD570PxBpmp6nfaEl48lzpUqpewPCU8lpEDqMsOeD26Z55Fa+s3HhnR/Aj6xZjV7rWrW4ke/wBMtLRGRLTaux1cuPnLlhtIAAUEE5490/ao8E6F4b8P3NxZeGksNW1z4h62Iby2t1/eC3ZoozJxwSiKQOhEZJ5JrqZtI8MftFfs56j8YI/hvo/gq1u9C0ywlbTbFoLa+1SOFWlgPGRmNiQzNkiPOflyf1WHFdark0K8UlUbkvJWel1vqmn5n5MuD6MM7qUJzbpxUXazvJSWtmuzTR8iaBoXirx54H1Hxo10mhx2V/8AZltpEE7ByAV8w8YLdRjAHAz1NcSNcXUtev8AwVrE0DajaQ+YZLY4jnXPJXrkjPPTPXGDXvXwi+F3hvwh8Ffip4i1a7GqX7a1b6fp0c84SKO3Wz3FiCdpHnSHDZ2/KAOtfP8A4v8Ah1qOoWul/Efwu4sL+wk8u2IkUi7JRhtwM5RgBkHpz0xXy+F4gzNY6Mqs3NTfvR6a6adj6bGcPZX/AGfKFKEYcqupemurufqP8c9Il/aA8BeHNS+HHh/7Ze634p0rUtMhsFWUJNdsFlR8HAAWWRXBPrmvGvi5+y/B+z1+0ZqHwW+PXx08LaL4VtJI75dOtPEksgfe8krskKIDCZWdCUkyU2MRuVga5X9lr9qAfsk/tCWPjTW9Ol1XwzpurtdXei27gBJTEyJcxBsDcu/OOAwBHBwa8S+JXxltPi3431/x34y8TT/2tqmpy3V1d6h8/mOzFiN46LkgDpgcdOK/HcPxo/qMalONqkn70ei816307I/Xa/C0XjXTrO8EtJdfR/cUfiz4m8aaBJefD/4NXcd4CJ7f/hI7bVXitZ4HkPzSQMvMxRUUTAEqFGCM4EPw/wDAsnhjwZEmrzte6ipKmWeYSbAV+bYTjAJBPTgNjNbOk+HYdL0qPxPcWsN5BK2FaO43q3fHUH8eld34kku/hP4LsvHPiPRtPvZNSkU6fYSFStvED8zsqjLsq5IBwOB1Jri/4iFn2BxSq0Yxulyq8eZLp1f4ms+BMkx2FlRquVm7u0uVvrbRH//Z\"\n", + " }\n", + " }\n", + " ]\n", + " ]\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Xq2Faefidx9U" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_predict:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"prediction\"].predict(endpoint=endpoint_id, instances=instances_list)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "QU9j-Yijdx9V" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dQew71RQdx9V" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_predict:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"predictions\": [\n", + " [\n", + " 0.0441863947,\n", + " 0.0965465382,\n", + " 0.131534964,\n", + " 0.111121736,\n", + " 0.133242667,\n", + " 0.0896093696,\n", + " 0.160808861,\n", + " 0.116257407,\n", + " 0.0309255011,\n", + " 0.0857665\n", + " ]\n", + " ],\n", + " \"deployedModelId\": \"1297564458264035328\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new" + }, + "source": [ + "### [projects.locations.endpoints.undeployModel](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/undeployModel)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "IYlo41cBdx9W" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_undeploymodel:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint_id, deployed_model_id=deployed_model_id, traffic_split={}\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "8_e5NspVdx9X" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "MdeQNL7idx9X" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "iha0pqA2TBDu" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup:migration,new" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "G5pYm8a4dx9Y" + }, + "outputs": [], + "source": [ + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_custom_job = True\n", + "delete_batchjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the model using the Vertex AI fully qualified identifier for the model\n", + "try:\n", + " if delete_model:\n", + " clients[\"model\"].delete_model(name=model_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex AI fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint:\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom training using the Vertex AI fully qualified identifier for the custom training\n", + "try:\n", + " if delete_custom_job:\n", + " clients[\"job\"].delete_custom_job(name=custom_training_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex AI fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob:\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r gs://$BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "collapsed_sections": [ + "gcp_authenticate", + "bucket:batch_prediction", + "setup_vars", + "import_aip", + "aip_constants", + "endpoints_predict:migration,new" + ], + "name": "UJ3 unified Custom Training Custom Container (TF Keras).ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/migration/UJ4 legacy AutoML Tables Regression.ipynb b/notebooks/community/migration/UJ4 legacy AutoML Tables Regression.ipynb new file mode 100644 index 000000000..2eccc3b65 --- /dev/null +++ b/notebooks/community/migration/UJ4 legacy AutoML Tables Regression.ipynb @@ -0,0 +1,10233 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "0YEWQgiBU7gh" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License.\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_title:migration,automl,icn" + }, + "source": [ + "# AutoML SDK: AutoML tables regression model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "JE-aKjayJjkF" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of AutoML SDK." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "7dxB6-B0JjkG" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-automl\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Iddo-8HmJjkH" + }, + "source": [ + "Install the Google *cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "xHvi6EFSJjkH" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-storage\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ktRTgB8DJjkI" + }, + "source": [ + "### Restart the Kernel\n", + "\n", + "Once you've installed the AutoML SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "x1g4mBdlJjkI" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "\n", + "if not os.getenv(\"AUTORUN\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)\n", + " " + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pxAMbzEgfD7e" + }, + "source": [ + "## Before you begin\r\n", + "\r\n", + "### GPU run-time\r\n", + "\r\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\r\n", + "\r\n", + "### Set up your GCP project\r\n", + "\r\n", + "**The following steps are required, regardless of your notebook environment.**\r\n", + "\r\n", + "1. [Select or create a GCP project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\r\n", + "\r\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\r\n", + "\r\n", + "3. [Enable the AutoML APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\r\n", + "\r\n", + "4. [Google Cloud SDK](https://cloud.google.com/sdk) is already installed in AutoML Notebooks.\r\n", + "\r\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\r\n", + "Cloud SDK uses the right project for all the commands in this notebook.\r\n", + "\r\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands.\r\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "hceMthWGJjkI" + }, + "source": [ + "#### Project ID\n", + "\n", + "**If you don't know your project ID**, try to get your project ID using `gcloud` command by executing the second cell below." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" #@param {type:\"string\"}\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id", + "outputId": "2663b1ca-f6c6-43ed-c23b-654919cb200c" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "0UfsdLCEJjkJ" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "jT0fijejJjkK" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for AutoML. We recommend when possible, to choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You cannot use a Multi-Regional Storage bucket for training with AutoML. Not all regions provide support for all AutoML services. For the latest support per region, see [Region support for AutoML services]()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "CTBlncfrJjkK" + }, + "outputs": [], + "source": [ + "REGION = 'us-central1' #@param {type: \"string\"}\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "J_croPBoJjkK" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "KKxStk5bJjkL" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "R0InH8s1f-uL" + }, + "source": [ + "### Authenticate your GCP account\r\n", + "\r\n", + "**If you are using AutoML Notebooks**, your environment is already\r\n", + "authenticated. Skip this step.\r\n", + "\r\n", + "*Note: If you are on an AutoML notebook and run the cell, the cell knows to skip executing the authentication steps.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "6aUuFIftxXiN" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your Google Cloud account. This provides access\n", + "# to your Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on AutoML, then don't execute this code\n", + "if not os.path.exists('/opt/deeplearning/metadata/env_version'):\n", + " if 'google.colab' in sys.modules:\n", + " from google.colab import auth as google_auth\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this tutorial in a notebook locally, replace the string\n", + " # below with the path to your service account key and run this cell to\n", + " # authenticate your Google Cloud account.\n", + " else:\n", + " %env GOOGLE_APPLICATION_CREDENTIALS your_path_to_credentials.json\n", + "\n", + " # Log in to your account on Google Cloud\n", + " ! gcloud auth login\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "MDUAZaN3JjkL" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"[your-bucket-name]\" #@param {type:\"string\"}\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n", + " BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "yU6VuylGJjkM" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "QVRixM4oJjkM" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION gs://$BUCKET_NAME\n", + " " + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Mi-Pdh6QJjkN" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aqDptKpYJjkN" + }, + "outputs": [], + "source": [ + "! gsutil ls -al gs://$BUCKET_NAME\n", + " " + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "fR9geV9pJjkO" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "hIHTX-pkJjkO" + }, + "source": [ + "#### Import AutoML SDK\n", + "\n", + "Import the AutoM SDK into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "zbd7zoxhU7he" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "\n", + "from google.cloud import automl_v1beta1 as automl\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Value\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "lbv411XjJjkP" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Setup up the following constants for AutoML:\n", + "\n", + "- `PARENT`: The AutoML location root path for dataset, model and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "rG1L9BEzxXiR" + }, + "outputs": [], + "source": [ + "# AutoML location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Rw-wcM5qU7h0" + }, + "source": [ + "## Clients\n", + "\n", + "The AutoML SDK works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the server (AutoML).\n", + "\n", + "You will use several clients in this tutorial, so set them all up upfront.\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "c3sR3OxmxXiR", + "outputId": "5bc213e7-2df8-4aca-dbf8-f8d267e98d52" + }, + "outputs": [], + "source": [ + "def automl_client():\n", + " return automl.AutoMlClient()\n", + "\n", + "\n", + "def operations_client():\n", + " return automl.AutoMlClient()._transport.operations_client\n", + "\n", + "\n", + "def prediction_client():\n", + " return automl.PredictionServiceClient()\n", + "\n", + "\n", + "def tables_client():\n", + " return automl.TablesClient(project=PROJECT_ID, region=REGION)\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"automl\"] = automl_client()\n", + "clients[\"operations\"] = operations_client()\n", + "clients[\"prediction\"] = prediction_client()\n", + "clients[\"tables\"] = tables_client()\n", + "\n", + "\n", + "for client in clients.items():\n", + " print(client)\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:flowers,csv,icn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = 'gs://cloud-ml-tables-data/bank-marketing.csv'\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "KHJnNxI5xXiS", + "outputId": "7714d066-2301-4b47-d155-d3c5d454def7" + }, + "outputs": [], + "source": [ + "! gsutil cat $IMPORT_FILE | head -n 10 \n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_create:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "Age,Job,MaritalStatus,Education,Default,Balance,Housing,Loan,Contact,Day,Month,Duration,Campaign,PDays,Previous,POutcome,Deposit\n", + "58,management,married,tertiary,no,2143,yes,no,unknown,5,may,261,1,-1,0,unknown,1\n", + "44,technician,single,secondary,no,29,yes,no,unknown,5,may,151,1,-1,0,unknown,1\n", + "33,entrepreneur,married,secondary,no,2,yes,yes,unknown,5,may,76,1,-1,0,unknown,1\n", + "47,blue-collar,married,unknown,no,1506,yes,no,unknown,5,may,92,1,-1,0,unknown,1\n", + "33,unknown,single,unknown,no,1,no,no,unknown,5,may,198,1,-1,0,unknown,1\n", + "35,management,married,tertiary,no,231,yes,no,unknown,5,may,139,1,-1,0,unknown,1\n", + "28,management,single,tertiary,no,447,yes,yes,unknown,5,may,217,1,-1,0,unknown,1\n", + "42,entrepreneur,divorced,tertiary,yes,2,yes,no,unknown,5,may,380,1,-1,0,unknown,1\n", + "58,retired,married,primary,no,121,yes,no,unknown,5,may,50,1,-1,0,unknown,1\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_create_dataset:migration" + }, + "source": [ + "## Create a dataset" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "eNQ_vjVFU7h8" + }, + "source": [ + "### Prepare data" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_datasets_create:migration,old" + }, + "source": [ + "### [projects.locations.datasets.create](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.datasets/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ozaMOwdNJjkT" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Rm2MLKeBU7iD", + "outputId": "2100960c-bfa5-42ba-a27c-d2047d106e10" + }, + "outputs": [], + "source": [ + "print(MessageToJson(\n", + " automl.CreateDatasetRequest(\n", + " dataset=automl.Dataset(\n", + " display_name=\"bank_\" + TIMESTAMP\n", + " )\n", + " ).__dict__[\"_pb\"])\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"dataset\": {\n", + " \"displayName\": \"bank_20210228161105\"\n", + " }\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "vHkuSZxkJjkT" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "SucGG7NhU7iE" + }, + "outputs": [], + "source": [ + "request = clients[\"tables\"].create_dataset(\n", + " dataset_display_name=\"bank_\" + TIMESTAMP\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "FLqkZMD2JjkT" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "_XZiIFP4U7iT", + "outputId": "f601b5e0-5f3d-438d-d942-e17e8f7360b8" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/datasets/TBL3707245895360708608\",\n", + " \"displayName\": \"bank_20210228161105\",\n", + " \"createTime\": \"2021-02-28T16:11:11.629918Z\",\n", + " \"etag\": \"AB3BwFrEEPSvg_CecNSkJ7f0ROXZnZxzgAD5hxVGnRz7NlK5a_WrkIfS0tGbkJTeYFpQ\",\n", + " \"tablesDatasetMetadata\": {\n", + " \"statsUpdateTime\": \"1970-01-01T00:00:00Z\"\n", + " }\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = request.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split('/')[-1]\n", + "\n", + "print(dataset_id)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_datasets_importdata:migration,old" + }, + "source": [ + "### [projects.locations.datasets.importData](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.datasets/importData)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "nacQ0lvTU7iU" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "6F894YodU7iZ", + "outputId": "25becc2c-04f4-4dc8-e8d4-3c0d1bae1122" + }, + "outputs": [], + "source": [ + "print(MessageToJson(\n", + " automl.ImportDataRequest(\n", + " name=dataset_id,\n", + " input_config={\"gcs_source\": {\"input_uris\": [IMPORT_FILE]}}\n", + " ).__dict__[\"_pb\"])\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/datasets/TBL3707245895360708608\",\n", + " \"inputConfig\": {\n", + " \"gcsSource\": {\n", + " \"inputUris\": [\n", + " \"gs://cloud-ml-tables-data/bank-marketing.csv\"\n", + " ]\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "NEE3oQPOU7ia" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "TitFNA4FU7ia" + }, + "outputs": [], + "source": [ + "request = clients[\"tables\"].import_data(\n", + " dataset_name=dataset_id,\n", + " gcs_input_uris=[IMPORT_FILE]\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "BfEBJzZ_U7ib" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Z5_f9IT5U7ib", + "outputId": "c061c7b5-9da8-47a2-b084-6aee028e428d" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "-zzom8TCkTzC" + }, + "source": [ + "*Example output*:\r\n", + "```\r\n", + "{}\r\n", + "```" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "zYDxpb87U7ig" + }, + "source": [ + "### [projects.locations.datasets.patch](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.datasets/patch)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "DS-tkN-lU7ig" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "TRGKYwFWU7ih" + }, + "outputs": [], + "source": [ + "request = clients[\"tables\"].set_target_column(\n", + " dataset_name=dataset_id,\n", + " column_spec_display_name='Deposit'\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "LoBmmTiHU7ii" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Kfji8sSjU7im", + "outputId": "99544fd6-95d5-43fe-de72-67ec3a921f58" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/datasets/TBL3707245895360708608\",\n", + " \"displayName\": \"bank_20210228161105\",\n", + " \"createTime\": \"2021-02-28T16:11:11.629918Z\",\n", + " \"etag\": \"AB3BwFoGe9VDggahb_dNWX28GjKcqW9XDUytMYvTDIoFWTv2pbmfLrpMB2kge33oW2MY\",\n", + " \"exampleCount\": 45211,\n", + " \"tablesDatasetMetadata\": {\n", + " \"primaryTableSpecId\": \"7767359156935196672\",\n", + " \"targetColumnSpecId\": \"2361404080544284672\",\n", + " \"statsUpdateTime\": \"2021-02-28T16:11:54.368477Z\"\n", + " }\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "uFP7B00cU7in" + }, + "source": [ + "## Train a model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_models_create:migration,old" + }, + "source": [ + "### [projects.locations.models.create](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wX-bPrRTU7io" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dYN44cllU7io" + }, + "outputs": [], + "source": [ + "request = clients[\"tables\"].create_model(\n", + " model_display_name=\"bank_\" + TIMESTAMP,\n", + " dataset_name=dataset_id,\n", + " train_budget_milli_node_hours=1*1000\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "GJu9d2CFU7it" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "_4frkgiOU7it", + "outputId": "737dfa45-7623-4d16-cec0-4b77fd781c8c" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/TBL8310014324731019264\"\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "# The full unique ID for the training pipeline\n", + "model_id = result.name\n", + "# The short numeric ID for the training pipeline\n", + "model_short_id = model_id.split('/')[-1]\n", + "\n", + "print(model_id)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_evaluate_the_model:migration" + }, + "source": [ + "## Evaluate the model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "yfufMwAEJjkX" + }, + "source": [ + "### [projects.locations.models.modelEvaluations.list](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models.modelEvaluations/list)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "SjYvkggyU7iy" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wI1d4wFsU7iz" + }, + "outputs": [], + "source": [ + "request = clients[\"tables\"].list_model_evaluations(\n", + " model_name=model_id\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "jFRv00YOU7iz" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "DwVP9IIiU7i1", + "outputId": "2b212216-331b-40b3-fecb-8e497ca1e52b", + "scrolled": true + }, + "outputs": [], + "source": [ + "evaluations_list = [\n", + " json.loads(MessageToJson(me.__dict__[\"_pb\"])) \n", + " for me in request.model_evaluation\n", + "]\n", + "\n", + "print(json.dumps(evaluations_list, indent=2))\n", + "# The evaluation slice\n", + "evaluation_slice = request.model_evaluation[0].name\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "[\n", + " {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/TBL8310014324731019264/modelEvaluations/124023939996839080\",\n", + " \"annotationSpecId\": \"not available\",\n", + " \"createTime\": \"2021-02-28T17:24:31.800758Z\",\n", + " \"evaluatedExampleCount\": 4639,\n", + " \"classificationEvaluationMetrics\": {\n", + " \"auPrc\": 0.89604473,\n", + " \"confidenceMetricsEntry\": [\n", + " {\n", + " \"recall\": 0.90515196,\n", + " \"precision\": 0.90515196,\n", + " \"f1Score\": 0.90515196,\n", + " \"falsePositiveRate\": 0.09484803,\n", + " \"truePositiveCount\": \"4199\",\n", + " \"falsePositiveCount\": \"440\",\n", + " \"falseNegativeCount\": \"440\",\n", + " \"trueNegativeCount\": \"4199\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.009267787,\n", + " \"recall\": 0.90515196,\n", + " \"precision\": 0.90515196,\n", + " \"f1Score\": 0.90515196,\n", + " \"falsePositiveRate\": 0.09484803,\n", + " \"truePositiveCount\": \"4199\",\n", + " \"falsePositiveCount\": \"440\",\n", + " \"falseNegativeCount\": \"440\",\n", + " \"trueNegativeCount\": \"4199\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.08771089,\n", + " \"recall\": 0.90515196,\n", + " \"precision\": 0.90515196,\n", + " \"f1Score\": 0.90515196,\n", + " \"falsePositiveRate\": 0.09484803,\n", + " \"truePositiveCount\": \"4199\",\n", + " \"falsePositiveCount\": \"440\",\n", + " \"falseNegativeCount\": \"440\",\n", + " \"trueNegativeCount\": \"4199\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.18867286,\n", + " \"recall\": 0.90515196,\n", + " \"precision\": 0.90515196,\n", + " \"f1Score\": 0.90515196,\n", + " \"falsePositiveRate\": 0.09484803,\n", + " \"truePositiveCount\": \"4199\",\n", + " \"falsePositiveCount\": \"440\",\n", + " \"falseNegativeCount\": \"440\",\n", + " \"trueNegativeCount\": \"4199\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.24711765,\n", + " \"recall\": 0.90515196,\n", + " \"precision\": 0.90515196,\n", + " \"f1Score\": 0.90515196,\n", + " \"falsePositiveRate\": 0.09484803,\n", + " \"truePositiveCount\": \"4199\",\n", + " \"falsePositiveCount\": \"440\",\n", + " \"falseNegativeCount\": \"440\",\n", + " \"trueNegativeCount\": \"4199\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.29667678,\n", + " \"recall\": 0.90515196,\n", + " \"precision\": 0.90515196,\n", + " \"f1Score\": 0.90515196,\n", + " \"falsePositiveRate\": 0.09484803,\n", + " \"truePositiveCount\": \"4199\",\n", + " \"falsePositiveCount\": \"440\",\n", + " \"falseNegativeCount\": \"440\",\n", + " \"trueNegativeCount\": \"4199\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.33910435,\n", + " \"recall\": 0.90515196,\n", + " \"precision\": 0.90515196,\n", + " \"f1Score\": 0.90515196,\n", + " \"falsePositiveRate\": 0.09484803,\n", + " \"truePositiveCount\": \"4199\",\n", + " \"falsePositiveCount\": \"440\",\n", + " \"falseNegativeCount\": \"440\",\n", + " \"trueNegativeCount\": \"4199\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.3722213,\n", + " \"recall\": 0.90515196,\n", + " \"precision\": 0.90515196,\n", + " \"f1Score\": 0.90515196,\n", + " \"falsePositiveRate\": 0.09484803,\n", + " \"truePositiveCount\": \"4199\",\n", + " \"falsePositiveCount\": \"440\",\n", + " \"falseNegativeCount\": \"440\",\n", + " \"trueNegativeCount\": \"4199\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.41141066,\n", + " \"recall\": 0.90515196,\n", + " \"precision\": 0.90515196,\n", + " \"f1Score\": 0.90515196,\n", + " \"falsePositiveRate\": 0.09484803,\n", + " \"truePositiveCount\": \"4199\",\n", + " \"falsePositiveCount\": \"440\",\n", + " \"falseNegativeCount\": \"440\",\n", + " \"trueNegativeCount\": \"4199\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.44597256,\n", + " \"recall\": 0.90515196,\n", + " \"precision\": 0.90515196,\n", + " \"f1Score\": 0.90515196,\n", + " \"falsePositiveRate\": 0.09484803,\n", + " \"truePositiveCount\": \"4199\",\n", + " \"falsePositiveCount\": \"440\",\n", + " \"falseNegativeCount\": \"440\",\n", + " \"trueNegativeCount\": \"4199\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.48088387,\n", + " \"recall\": 0.90515196,\n", + " \"precision\": 0.90515196,\n", + " \"f1Score\": 0.90515196,\n", + " \"falsePositiveRate\": 0.09484803,\n", + " \"truePositiveCount\": \"4199\",\n", + " \"falsePositiveCount\": \"440\",\n", + " \"falseNegativeCount\": \"440\",\n", + " \"trueNegativeCount\": \"4199\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5,\n", + " \"recall\": 0.90515196,\n", + " \"precision\": 0.90515196,\n", + " \"f1Score\": 0.90515196,\n", + " \"falsePositiveRate\": 0.09484803,\n", + " \"truePositiveCount\": \"4199\",\n", + " \"falsePositiveCount\": \"440\",\n", + " \"falseNegativeCount\": \"440\",\n", + " \"trueNegativeCount\": \"4199\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.51082355,\n", + " \"recall\": 0.9006251,\n", + " \"precision\": 0.90688086,\n", + " \"f1Score\": 0.90374213,\n", + " \"falsePositiveRate\": 0.09247683,\n", + " \"truePositiveCount\": \"4178\",\n", + " \"falsePositiveCount\": \"429\",\n", + " \"falseNegativeCount\": \"461\",\n", + " \"trueNegativeCount\": \"4210\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5395309,\n", + " \"recall\": 0.8902781,\n", + " \"precision\": 0.9145261,\n", + " \"f1Score\": 0.9022392,\n", + " \"falsePositiveRate\": 0.083207585,\n", + " \"truePositiveCount\": \"4130\",\n", + " \"falsePositiveCount\": \"386\",\n", + " \"falseNegativeCount\": \"509\",\n", + " \"trueNegativeCount\": \"4253\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.569127,\n", + " \"recall\": 0.88014656,\n", + " \"precision\": 0.92083895,\n", + " \"f1Score\": 0.90003306,\n", + " \"falsePositiveRate\": 0.07566286,\n", + " \"truePositiveCount\": \"4083\",\n", + " \"falsePositiveCount\": \"351\",\n", + " \"falseNegativeCount\": \"556\",\n", + " \"trueNegativeCount\": \"4288\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.59819055,\n", + " \"recall\": 0.8700151,\n", + " \"precision\": 0.9284564,\n", + " \"f1Score\": 0.8982862,\n", + " \"falsePositiveRate\": 0.06704031,\n", + " \"truePositiveCount\": \"4036\",\n", + " \"falsePositiveCount\": \"311\",\n", + " \"falseNegativeCount\": \"603\",\n", + " \"trueNegativeCount\": \"4328\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6267493,\n", + " \"recall\": 0.86009914,\n", + " \"precision\": 0.9350832,\n", + " \"f1Score\": 0.8960252,\n", + " \"falsePositiveRate\": 0.059711143,\n", + " \"truePositiveCount\": \"3990\",\n", + " \"falsePositiveCount\": \"277\",\n", + " \"falseNegativeCount\": \"649\",\n", + " \"trueNegativeCount\": \"4362\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6483357,\n", + " \"recall\": 0.84996766,\n", + " \"precision\": 0.9401526,\n", + " \"f1Score\": 0.8927884,\n", + " \"falsePositiveRate\": 0.05410649,\n", + " \"truePositiveCount\": \"3943\",\n", + " \"falsePositiveCount\": \"251\",\n", + " \"falseNegativeCount\": \"696\",\n", + " \"trueNegativeCount\": \"4388\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.68742466,\n", + " \"recall\": 0.8404829,\n", + " \"precision\": 0.95051193,\n", + " \"f1Score\": 0.8921176,\n", + " \"falsePositiveRate\": 0.04375943,\n", + " \"truePositiveCount\": \"3899\",\n", + " \"falsePositiveCount\": \"203\",\n", + " \"falseNegativeCount\": \"740\",\n", + " \"trueNegativeCount\": \"4436\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.7178758,\n", + " \"recall\": 0.83035135,\n", + " \"precision\": 0.95630586,\n", + " \"f1Score\": 0.8888889,\n", + " \"falsePositiveRate\": 0.03793921,\n", + " \"truePositiveCount\": \"3852\",\n", + " \"falsePositiveCount\": \"176\",\n", + " \"falseNegativeCount\": \"787\",\n", + " \"trueNegativeCount\": \"4463\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.7454512,\n", + " \"recall\": 0.8200043,\n", + " \"precision\": 0.9620637,\n", + " \"f1Score\": 0.8853718,\n", + " \"falsePositiveRate\": 0.032334555,\n", + " \"truePositiveCount\": \"3804\",\n", + " \"falsePositiveCount\": \"150\",\n", + " \"falseNegativeCount\": \"835\",\n", + " \"trueNegativeCount\": \"4489\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.7702336,\n", + " \"recall\": 0.8105195,\n", + " \"precision\": 0.9685729,\n", + " \"f1Score\": 0.8825255,\n", + " \"falsePositiveRate\": 0.02629877,\n", + " \"truePositiveCount\": \"3760\",\n", + " \"falsePositiveCount\": \"122\",\n", + " \"falseNegativeCount\": \"879\",\n", + " \"trueNegativeCount\": \"4517\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.79181457,\n", + " \"recall\": 0.80038804,\n", + " \"precision\": 0.972244,\n", + " \"f1Score\": 0.87798536,\n", + " \"falsePositiveRate\": 0.022849752,\n", + " \"truePositiveCount\": \"3713\",\n", + " \"falsePositiveCount\": \"106\",\n", + " \"falseNegativeCount\": \"926\",\n", + " \"trueNegativeCount\": \"4533\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.80905926,\n", + " \"recall\": 0.7902565,\n", + " \"precision\": 0.9755189,\n", + " \"f1Score\": 0.873169,\n", + " \"falsePositiveRate\": 0.01983186,\n", + " \"truePositiveCount\": \"3666\",\n", + " \"falsePositiveCount\": \"92\",\n", + " \"falseNegativeCount\": \"973\",\n", + " \"trueNegativeCount\": \"4547\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.82475233,\n", + " \"recall\": 0.7803406,\n", + " \"precision\": 0.97574127,\n", + " \"f1Score\": 0.86716974,\n", + " \"falsePositiveRate\": 0.019400733,\n", + " \"truePositiveCount\": \"3620\",\n", + " \"falsePositiveCount\": \"90\",\n", + " \"falseNegativeCount\": \"1019\",\n", + " \"trueNegativeCount\": \"4549\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.84347624,\n", + " \"recall\": 0.7702091,\n", + " \"precision\": 0.97863597,\n", + " \"f1Score\": 0.86200243,\n", + " \"falsePositiveRate\": 0.01681397,\n", + " \"truePositiveCount\": \"3573\",\n", + " \"falsePositiveCount\": \"78\",\n", + " \"falseNegativeCount\": \"1066\",\n", + " \"trueNegativeCount\": \"4561\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.8612723,\n", + " \"recall\": 0.7602932,\n", + " \"precision\": 0.9799944,\n", + " \"f1Score\": 0.8562758,\n", + " \"falsePositiveRate\": 0.015520587,\n", + " \"truePositiveCount\": \"3527\",\n", + " \"falsePositiveCount\": \"72\",\n", + " \"falseNegativeCount\": \"1112\",\n", + " \"trueNegativeCount\": \"4567\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.87450475,\n", + " \"recall\": 0.75016165,\n", + " \"precision\": 0.9811108,\n", + " \"f1Score\": 0.8502321,\n", + " \"falsePositiveRate\": 0.014442768,\n", + " \"truePositiveCount\": \"3480\",\n", + " \"falsePositiveCount\": \"67\",\n", + " \"falseNegativeCount\": \"1159\",\n", + " \"trueNegativeCount\": \"4572\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.8876167,\n", + " \"recall\": 0.74003017,\n", + " \"precision\": 0.9825415,\n", + " \"f1Score\": 0.8442149,\n", + " \"falsePositiveRate\": 0.013149385,\n", + " \"truePositiveCount\": \"3433\",\n", + " \"falsePositiveCount\": \"61\",\n", + " \"falseNegativeCount\": \"1206\",\n", + " \"trueNegativeCount\": \"4578\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.89729506,\n", + " \"recall\": 0.7303298,\n", + " \"precision\": 0.98402554,\n", + " \"f1Score\": 0.8384063,\n", + " \"falsePositiveRate\": 0.011856004,\n", + " \"truePositiveCount\": \"3388\",\n", + " \"falsePositiveCount\": \"55\",\n", + " \"falseNegativeCount\": \"1251\",\n", + " \"trueNegativeCount\": \"4584\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9069939,\n", + " \"recall\": 0.72041386,\n", + " \"precision\": 0.9858407,\n", + " \"f1Score\": 0.8324823,\n", + " \"falsePositiveRate\": 0.010347057,\n", + " \"truePositiveCount\": \"3342\",\n", + " \"falsePositiveCount\": \"48\",\n", + " \"falseNegativeCount\": \"1297\",\n", + " \"trueNegativeCount\": \"4591\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.91544837,\n", + " \"recall\": 0.7102824,\n", + " \"precision\": 0.987118,\n", + " \"f1Score\": 0.8261251,\n", + " \"falsePositiveRate\": 0.009269239,\n", + " \"truePositiveCount\": \"3295\",\n", + " \"falsePositiveCount\": \"43\",\n", + " \"falseNegativeCount\": \"1344\",\n", + " \"trueNegativeCount\": \"4596\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9246296,\n", + " \"recall\": 0.700582,\n", + " \"precision\": 0.9896468,\n", + " \"f1Score\": 0.8203963,\n", + " \"falsePositiveRate\": 0.007329166,\n", + " \"truePositiveCount\": \"3250\",\n", + " \"falsePositiveCount\": \"34\",\n", + " \"falseNegativeCount\": \"1389\",\n", + " \"trueNegativeCount\": \"4605\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.932719,\n", + " \"recall\": 0.6900194,\n", + " \"precision\": 0.99071497,\n", + " \"f1Score\": 0.8134689,\n", + " \"falsePositiveRate\": 0.006466911,\n", + " \"truePositiveCount\": \"3201\",\n", + " \"falsePositiveCount\": \"30\",\n", + " \"falseNegativeCount\": \"1438\",\n", + " \"trueNegativeCount\": \"4609\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.93940556,\n", + " \"recall\": 0.6807502,\n", + " \"precision\": 0.9921458,\n", + " \"f1Score\": 0.80746615,\n", + " \"falsePositiveRate\": 0.0053890925,\n", + " \"truePositiveCount\": \"3158\",\n", + " \"falsePositiveCount\": \"25\",\n", + " \"falseNegativeCount\": \"1481\",\n", + " \"trueNegativeCount\": \"4614\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9436984,\n", + " \"recall\": 0.6704031,\n", + " \"precision\": 0.99234205,\n", + " \"f1Score\": 0.8002058,\n", + " \"falsePositiveRate\": 0.0051735286,\n", + " \"truePositiveCount\": \"3110\",\n", + " \"falsePositiveCount\": \"24\",\n", + " \"falseNegativeCount\": \"1529\",\n", + " \"trueNegativeCount\": \"4615\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9486017,\n", + " \"recall\": 0.6602716,\n", + " \"precision\": 0.992547,\n", + " \"f1Score\": 0.7930097,\n", + " \"falsePositiveRate\": 0.004957965,\n", + " \"truePositiveCount\": \"3063\",\n", + " \"falsePositiveCount\": \"23\",\n", + " \"falseNegativeCount\": \"1576\",\n", + " \"trueNegativeCount\": \"4616\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.95302504,\n", + " \"recall\": 0.6501401,\n", + " \"precision\": 0.9937397,\n", + " \"f1Score\": 0.78603077,\n", + " \"falsePositiveRate\": 0.0040957103,\n", + " \"truePositiveCount\": \"3016\",\n", + " \"falsePositiveCount\": \"19\",\n", + " \"falseNegativeCount\": \"1623\",\n", + " \"trueNegativeCount\": \"4620\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.95584136,\n", + " \"recall\": 0.6400086,\n", + " \"precision\": 0.99530673,\n", + " \"f1Score\": 0.7790606,\n", + " \"falsePositiveRate\": 0.003017892,\n", + " \"truePositiveCount\": \"2969\",\n", + " \"falsePositiveCount\": \"14\",\n", + " \"falseNegativeCount\": \"1670\",\n", + " \"trueNegativeCount\": \"4625\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.95881057,\n", + " \"recall\": 0.6300927,\n", + " \"precision\": 0.9955722,\n", + " \"f1Score\": 0.7717492,\n", + " \"falsePositiveRate\": 0.002802328,\n", + " \"truePositiveCount\": \"2923\",\n", + " \"falsePositiveCount\": \"13\",\n", + " \"falseNegativeCount\": \"1716\",\n", + " \"trueNegativeCount\": \"4626\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.96166563,\n", + " \"recall\": 0.62082344,\n", + " \"precision\": 0.9961951,\n", + " \"f1Score\": 0.76494026,\n", + " \"falsePositiveRate\": 0.0023712006,\n", + " \"truePositiveCount\": \"2880\",\n", + " \"falsePositiveCount\": \"11\",\n", + " \"falseNegativeCount\": \"1759\",\n", + " \"trueNegativeCount\": \"4628\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.96420497,\n", + " \"recall\": 0.61026084,\n", + " \"precision\": 0.9961295,\n", + " \"f1Score\": 0.75685066,\n", + " \"falsePositiveRate\": 0.0023712006,\n", + " \"truePositiveCount\": \"2831\",\n", + " \"falsePositiveCount\": \"11\",\n", + " \"falseNegativeCount\": \"1808\",\n", + " \"trueNegativeCount\": \"4628\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.96650803,\n", + " \"recall\": 0.5999138,\n", + " \"precision\": 0.996063,\n", + " \"f1Score\": 0.7488228,\n", + " \"falsePositiveRate\": 0.0023712006,\n", + " \"truePositiveCount\": \"2783\",\n", + " \"falsePositiveCount\": \"11\",\n", + " \"falseNegativeCount\": \"1856\",\n", + " \"trueNegativeCount\": \"4628\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.96897614,\n", + " \"recall\": 0.59042895,\n", + " \"precision\": 0.9963623,\n", + " \"f1Score\": 0.74147266,\n", + " \"falsePositiveRate\": 0.002155637,\n", + " \"truePositiveCount\": \"2739\",\n", + " \"falsePositiveCount\": \"10\",\n", + " \"falseNegativeCount\": \"1900\",\n", + " \"trueNegativeCount\": \"4629\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.97092766,\n", + " \"recall\": 0.5807286,\n", + " \"precision\": 0.99667037,\n", + " \"f1Score\": 0.73385996,\n", + " \"falsePositiveRate\": 0.0019400733,\n", + " \"truePositiveCount\": \"2694\",\n", + " \"falsePositiveCount\": \"9\",\n", + " \"falseNegativeCount\": \"1945\",\n", + " \"trueNegativeCount\": \"4630\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9727514,\n", + " \"recall\": 0.5699504,\n", + " \"precision\": 0.9969834,\n", + " \"f1Score\": 0.7252777,\n", + " \"falsePositiveRate\": 0.0017245096,\n", + " \"truePositiveCount\": \"2644\",\n", + " \"falsePositiveCount\": \"8\",\n", + " \"falseNegativeCount\": \"1995\",\n", + " \"trueNegativeCount\": \"4631\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9747563,\n", + " \"recall\": 0.56046563,\n", + " \"precision\": 0.99731493,\n", + " \"f1Score\": 0.7176373,\n", + " \"falsePositiveRate\": 0.001508946,\n", + " \"truePositiveCount\": \"2600\",\n", + " \"falsePositiveCount\": \"7\",\n", + " \"falseNegativeCount\": \"2039\",\n", + " \"trueNegativeCount\": \"4632\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9766838,\n", + " \"recall\": 0.549903,\n", + " \"precision\": 0.9976535,\n", + " \"f1Score\": 0.709005,\n", + " \"falsePositiveRate\": 0.0012933821,\n", + " \"truePositiveCount\": \"2551\",\n", + " \"falsePositiveCount\": \"6\",\n", + " \"falseNegativeCount\": \"2088\",\n", + " \"trueNegativeCount\": \"4633\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.97776866,\n", + " \"recall\": 0.54063374,\n", + " \"precision\": 0.99801034,\n", + " \"f1Score\": 0.7013423,\n", + " \"falsePositiveRate\": 0.0010778185,\n", + " \"truePositiveCount\": \"2508\",\n", + " \"falsePositiveCount\": \"5\",\n", + " \"falseNegativeCount\": \"2131\",\n", + " \"trueNegativeCount\": \"4634\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9792647,\n", + " \"recall\": 0.53050226,\n", + " \"precision\": 0.9979724,\n", + " \"f1Score\": 0.6927516,\n", + " \"falsePositiveRate\": 0.0010778185,\n", + " \"truePositiveCount\": \"2461\",\n", + " \"falsePositiveCount\": \"5\",\n", + " \"falseNegativeCount\": \"2178\",\n", + " \"trueNegativeCount\": \"4634\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9804138,\n", + " \"recall\": 0.5203708,\n", + " \"precision\": 0.99834573,\n", + " \"f1Score\": 0.6841434,\n", + " \"falsePositiveRate\": 0.0008622548,\n", + " \"truePositiveCount\": \"2414\",\n", + " \"falsePositiveCount\": \"4\",\n", + " \"falseNegativeCount\": \"2225\",\n", + " \"trueNegativeCount\": \"4635\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9814273,\n", + " \"recall\": 0.51045483,\n", + " \"precision\": 0.9991561,\n", + " \"f1Score\": 0.6757027,\n", + " \"falsePositiveRate\": 0.0004311274,\n", + " \"truePositiveCount\": \"2368\",\n", + " \"falsePositiveCount\": \"2\",\n", + " \"falseNegativeCount\": \"2271\",\n", + " \"trueNegativeCount\": \"4637\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98243207,\n", + " \"recall\": 0.5005389,\n", + " \"precision\": 0.9991394,\n", + " \"f1Score\": 0.6669539,\n", + " \"falsePositiveRate\": 0.0004311274,\n", + " \"truePositiveCount\": \"2322\",\n", + " \"falsePositiveCount\": \"2\",\n", + " \"falseNegativeCount\": \"2317\",\n", + " \"trueNegativeCount\": \"4637\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9835001,\n", + " \"recall\": 0.49019185,\n", + " \"precision\": 0.9995604,\n", + " \"f1Score\": 0.6577958,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"2274\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2365\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9844133,\n", + " \"recall\": 0.4804915,\n", + " \"precision\": 0.9995516,\n", + " \"f1Score\": 0.6490028,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"2229\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2410\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98528725,\n", + " \"recall\": 0.47079113,\n", + " \"precision\": 0.99954236,\n", + " \"f1Score\": 0.6400938,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"2184\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2455\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98630506,\n", + " \"recall\": 0.46001294,\n", + " \"precision\": 0.9995316,\n", + " \"f1Score\": 0.6300561,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"2134\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2505\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9871925,\n", + " \"recall\": 0.44988143,\n", + " \"precision\": 0.9995211,\n", + " \"f1Score\": 0.6204846,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"2087\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2552\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9877509,\n", + " \"recall\": 0.44082776,\n", + " \"precision\": 0.99951124,\n", + " \"f1Score\": 0.6118175,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"2045\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2594\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9885219,\n", + " \"recall\": 0.43048072,\n", + " \"precision\": 0.9994995,\n", + " \"f1Score\": 0.6017779,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"1997\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2642\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9892192,\n", + " \"recall\": 0.4199181,\n", + " \"precision\": 0.9994869,\n", + " \"f1Score\": 0.5913783,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"1948\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2691\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9897657,\n", + " \"recall\": 0.41000214,\n", + " \"precision\": 0.9994745,\n", + " \"f1Score\": 0.5814735,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"1902\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2737\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9902156,\n", + " \"recall\": 0.40051734,\n", + " \"precision\": 0.99946207,\n", + " \"f1Score\": 0.57186824,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"1858\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2781\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.990671,\n", + " \"recall\": 0.3901703,\n", + " \"precision\": 0.9994478,\n", + " \"f1Score\": 0.5612403,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"1810\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2829\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99106044,\n", + " \"recall\": 0.38025436,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.5509917,\n", + " \"truePositiveCount\": \"1764\",\n", + " \"falseNegativeCount\": \"2875\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99142385,\n", + " \"recall\": 0.36990732,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.5400472,\n", + " \"truePositiveCount\": \"1716\",\n", + " \"falseNegativeCount\": \"2923\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.991821,\n", + " \"recall\": 0.35999137,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.52940243,\n", + " \"truePositiveCount\": \"1670\",\n", + " \"falseNegativeCount\": \"2969\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99219143,\n", + " \"recall\": 0.35007545,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.5186013,\n", + " \"truePositiveCount\": \"1624\",\n", + " \"falseNegativeCount\": \"3015\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99265563,\n", + " \"recall\": 0.3401595,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.50764036,\n", + " \"truePositiveCount\": \"1578\",\n", + " \"falseNegativeCount\": \"3061\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9930375,\n", + " \"recall\": 0.33045915,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.49675956,\n", + " \"truePositiveCount\": \"1533\",\n", + " \"falseNegativeCount\": \"3106\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9932301,\n", + " \"recall\": 0.32054323,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.48547176,\n", + " \"truePositiveCount\": \"1487\",\n", + " \"falseNegativeCount\": \"3152\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99349546,\n", + " \"recall\": 0.31041172,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.47376212,\n", + " \"truePositiveCount\": \"1440\",\n", + " \"falseNegativeCount\": \"3199\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.993766,\n", + " \"recall\": 0.30006468,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.461615,\n", + " \"truePositiveCount\": \"1392\",\n", + " \"falseNegativeCount\": \"3247\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9940096,\n", + " \"recall\": 0.29014874,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.44979113,\n", + " \"truePositiveCount\": \"1346\",\n", + " \"falseNegativeCount\": \"3293\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9942971,\n", + " \"recall\": 0.27980167,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.4372579,\n", + " \"truePositiveCount\": \"1298\",\n", + " \"falseNegativeCount\": \"3341\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9945464,\n", + " \"recall\": 0.26988575,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.42505518,\n", + " \"truePositiveCount\": \"1252\",\n", + " \"falseNegativeCount\": \"3387\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.994734,\n", + " \"recall\": 0.2601854,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.41293192,\n", + " \"truePositiveCount\": \"1207\",\n", + " \"falseNegativeCount\": \"3432\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9949861,\n", + " \"recall\": 0.25026944,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.40034482,\n", + " \"truePositiveCount\": \"1161\",\n", + " \"falseNegativeCount\": \"3478\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99523896,\n", + " \"recall\": 0.2399224,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.38699582,\n", + " \"truePositiveCount\": \"1113\",\n", + " \"falseNegativeCount\": \"3526\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99545944,\n", + " \"recall\": 0.23043759,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.37456203,\n", + " \"truePositiveCount\": \"1069\",\n", + " \"falseNegativeCount\": \"3570\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9956529,\n", + " \"recall\": 0.22009054,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.36077738,\n", + " \"truePositiveCount\": \"1021\",\n", + " \"falseNegativeCount\": \"3618\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9958653,\n", + " \"recall\": 0.21039017,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.34764025,\n", + " \"truePositiveCount\": \"976\",\n", + " \"falseNegativeCount\": \"3663\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99604857,\n", + " \"recall\": 0.20025867,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.33369252,\n", + " \"truePositiveCount\": \"929\",\n", + " \"falseNegativeCount\": \"3710\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99617386,\n", + " \"recall\": 0.19012718,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.31950733,\n", + " \"truePositiveCount\": \"882\",\n", + " \"falseNegativeCount\": \"3757\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9963188,\n", + " \"recall\": 0.18042682,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.3056976,\n", + " \"truePositiveCount\": \"837\",\n", + " \"falseNegativeCount\": \"3802\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99644965,\n", + " \"recall\": 0.17029533,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.29102966,\n", + " \"truePositiveCount\": \"790\",\n", + " \"falseNegativeCount\": \"3849\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99657893,\n", + " \"recall\": 0.16059496,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.27674592,\n", + " \"truePositiveCount\": \"745\",\n", + " \"falseNegativeCount\": \"3894\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99672806,\n", + " \"recall\": 0.1502479,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.2612444,\n", + " \"truePositiveCount\": \"697\",\n", + " \"falseNegativeCount\": \"3942\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9968369,\n", + " \"recall\": 0.14033197,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.24612476,\n", + " \"truePositiveCount\": \"651\",\n", + " \"falseNegativeCount\": \"3988\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9969655,\n", + " \"recall\": 0.13041604,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.23073989,\n", + " \"truePositiveCount\": \"605\",\n", + " \"falseNegativeCount\": \"4034\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9970963,\n", + " \"recall\": 0.12028454,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.21473928,\n", + " \"truePositiveCount\": \"558\",\n", + " \"falseNegativeCount\": \"4081\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9972023,\n", + " \"recall\": 0.110799745,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.19949543,\n", + " \"truePositiveCount\": \"514\",\n", + " \"falseNegativeCount\": \"4125\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9973033,\n", + " \"recall\": 0.100668244,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.18292205,\n", + " \"truePositiveCount\": \"467\",\n", + " \"falseNegativeCount\": \"4172\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99741936,\n", + " \"recall\": 0.09053675,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.16604072,\n", + " \"truePositiveCount\": \"420\",\n", + " \"falseNegativeCount\": \"4219\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9974875,\n", + " \"recall\": 0.080620825,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.14921205,\n", + " \"truePositiveCount\": \"374\",\n", + " \"falseNegativeCount\": \"4265\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9976044,\n", + " \"recall\": 0.070273764,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.13131924,\n", + " \"truePositiveCount\": \"326\",\n", + " \"falseNegativeCount\": \"4313\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99769753,\n", + " \"recall\": 0.06014227,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.11346076,\n", + " \"truePositiveCount\": \"279\",\n", + " \"falseNegativeCount\": \"4360\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99778664,\n", + " \"recall\": 0.049795214,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.09486653,\n", + " \"truePositiveCount\": \"231\",\n", + " \"falseNegativeCount\": \"4408\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9978742,\n", + " \"recall\": 0.040525977,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.07789517,\n", + " \"truePositiveCount\": \"188\",\n", + " \"falseNegativeCount\": \"4451\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.997992,\n", + " \"recall\": 0.030394481,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.058995817,\n", + " \"truePositiveCount\": \"141\",\n", + " \"falseNegativeCount\": \"4498\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9980989,\n", + " \"recall\": 0.020262988,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.03972111,\n", + " \"truePositiveCount\": \"94\",\n", + " \"falseNegativeCount\": \"4545\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99821925,\n", + " \"recall\": 0.010347057,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.020482184,\n", + " \"truePositiveCount\": \"48\",\n", + " \"falseNegativeCount\": \"4591\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9987167,\n", + " \"precision\": \"NaN\",\n", + " \"f1Score\": \"NaN\",\n", + " \"falseNegativeCount\": \"4639\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 1.0,\n", + " \"precision\": \"NaN\",\n", + " \"f1Score\": \"NaN\",\n", + " \"falseNegativeCount\": \"4639\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " }\n", + " ],\n", + " \"confusionMatrix\": {\n", + " \"row\": [\n", + " {\n", + " \"exampleCount\": [\n", + " 3941,\n", + " 147\n", + " ]\n", + " },\n", + " {\n", + " \"exampleCount\": [\n", + " 293,\n", + " 258\n", + " ]\n", + " }\n", + " ],\n", + " \"displayName\": [\n", + " \"1\",\n", + " \"2\"\n", + " ]\n", + " },\n", + " \"auRoc\": 0.94022393,\n", + " \"logLoss\": 0.20077285\n", + " }\n", + " },\n", + " {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/TBL8310014324731019264/modelEvaluations/542606150386485790\",\n", + " \"annotationSpecId\": \"not available\",\n", + " \"createTime\": \"2021-02-28T17:24:31.800758Z\",\n", + " \"evaluatedExampleCount\": 4088,\n", + " \"classificationEvaluationMetrics\": {\n", + " \"auPrc\": 0.9907691,\n", + " \"confidenceMetricsEntry\": [\n", + " {\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.8812244,\n", + " \"f1Score\": 0.9368626,\n", + " \"falsePositiveRate\": 1.0,\n", + " \"truePositiveCount\": \"4088\",\n", + " \"falsePositiveCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.07797722,\n", + " \"recall\": 0.9997554,\n", + " \"precision\": 0.8815789,\n", + " \"f1Score\": 0.9369555,\n", + " \"falsePositiveRate\": 0.99637026,\n", + " \"truePositiveCount\": \"4087\",\n", + " \"falsePositiveCount\": \"549\",\n", + " \"falseNegativeCount\": \"1\",\n", + " \"trueNegativeCount\": \"2\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.30998033,\n", + " \"recall\": 0.9899706,\n", + " \"precision\": 0.90254235,\n", + " \"f1Score\": 0.94423705,\n", + " \"falsePositiveRate\": 0.79310346,\n", + " \"truePositiveCount\": \"4047\",\n", + " \"falsePositiveCount\": \"437\",\n", + " \"falseNegativeCount\": \"41\",\n", + " \"trueNegativeCount\": \"114\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.39478266,\n", + " \"recall\": 0.9801859,\n", + " \"precision\": 0.913797,\n", + " \"f1Score\": 0.9458279,\n", + " \"falsePositiveRate\": 0.6860254,\n", + " \"truePositiveCount\": \"4007\",\n", + " \"falsePositiveCount\": \"378\",\n", + " \"falseNegativeCount\": \"81\",\n", + " \"trueNegativeCount\": \"173\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.46149504,\n", + " \"recall\": 0.96991193,\n", + " \"precision\": 0.9246735,\n", + " \"f1Score\": 0.9467526,\n", + " \"falsePositiveRate\": 0.5862069,\n", + " \"truePositiveCount\": \"3965\",\n", + " \"falsePositiveCount\": \"323\",\n", + " \"falseNegativeCount\": \"123\",\n", + " \"trueNegativeCount\": \"228\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5,\n", + " \"recall\": 0.9640411,\n", + " \"precision\": 0.9307983,\n", + " \"f1Score\": 0.9471281,\n", + " \"falsePositiveRate\": 0.53176045,\n", + " \"truePositiveCount\": \"3941\",\n", + " \"falsePositiveCount\": \"293\",\n", + " \"falseNegativeCount\": \"147\",\n", + " \"trueNegativeCount\": \"258\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.51386535,\n", + " \"recall\": 0.95988256,\n", + " \"precision\": 0.93361884,\n", + " \"f1Score\": 0.94656855,\n", + " \"falsePositiveRate\": 0.50635207,\n", + " \"truePositiveCount\": \"3924\",\n", + " \"falsePositiveCount\": \"279\",\n", + " \"falseNegativeCount\": \"164\",\n", + " \"trueNegativeCount\": \"272\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5659395,\n", + " \"recall\": 0.94985324,\n", + " \"precision\": 0.9406492,\n", + " \"f1Score\": 0.9452288,\n", + " \"falsePositiveRate\": 0.4446461,\n", + " \"truePositiveCount\": \"3883\",\n", + " \"falsePositiveCount\": \"245\",\n", + " \"falseNegativeCount\": \"205\",\n", + " \"trueNegativeCount\": \"306\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6215324,\n", + " \"recall\": 0.93982387,\n", + " \"precision\": 0.94817376,\n", + " \"f1Score\": 0.94398034,\n", + " \"falsePositiveRate\": 0.38112524,\n", + " \"truePositiveCount\": \"3842\",\n", + " \"falsePositiveCount\": \"210\",\n", + " \"falseNegativeCount\": \"246\",\n", + " \"trueNegativeCount\": \"341\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.65884733,\n", + " \"recall\": 0.93003917,\n", + " \"precision\": 0.9533601,\n", + " \"f1Score\": 0.9415552,\n", + " \"falsePositiveRate\": 0.33756804,\n", + " \"truePositiveCount\": \"3802\",\n", + " \"falsePositiveCount\": \"186\",\n", + " \"falseNegativeCount\": \"286\",\n", + " \"trueNegativeCount\": \"365\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.7143411,\n", + " \"recall\": 0.9200098,\n", + " \"precision\": 0.9623849,\n", + " \"f1Score\": 0.9407204,\n", + " \"falsePositiveRate\": 0.26678765,\n", + " \"truePositiveCount\": \"3761\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"327\",\n", + " \"trueNegativeCount\": \"404\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.7492718,\n", + " \"recall\": 0.9099804,\n", + " \"precision\": 0.96749026,\n", + " \"f1Score\": 0.9378545,\n", + " \"falsePositiveRate\": 0.22686026,\n", + " \"truePositiveCount\": \"3720\",\n", + " \"falsePositiveCount\": \"125\",\n", + " \"falseNegativeCount\": \"368\",\n", + " \"trueNegativeCount\": \"426\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.7795446,\n", + " \"recall\": 0.9001957,\n", + " \"precision\": 0.97303015,\n", + " \"f1Score\": 0.93519694,\n", + " \"falsePositiveRate\": 0.18511796,\n", + " \"truePositiveCount\": \"3680\",\n", + " \"falsePositiveCount\": \"102\",\n", + " \"falseNegativeCount\": \"408\",\n", + " \"trueNegativeCount\": \"449\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.7989856,\n", + " \"recall\": 0.8899217,\n", + " \"precision\": 0.97559667,\n", + " \"f1Score\": 0.93079185,\n", + " \"falsePositiveRate\": 0.16515426,\n", + " \"truePositiveCount\": \"3638\",\n", + " \"falsePositiveCount\": \"91\",\n", + " \"falseNegativeCount\": \"450\",\n", + " \"trueNegativeCount\": \"460\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.82042044,\n", + " \"recall\": 0.87989235,\n", + " \"precision\": 0.9777113,\n", + " \"f1Score\": 0.9262263,\n", + " \"falsePositiveRate\": 0.14882033,\n", + " \"truePositiveCount\": \"3597\",\n", + " \"falsePositiveCount\": \"82\",\n", + " \"falseNegativeCount\": \"491\",\n", + " \"trueNegativeCount\": \"469\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.83562684,\n", + " \"recall\": 0.87010765,\n", + " \"precision\": 0.9790807,\n", + " \"f1Score\": 0.92138326,\n", + " \"falsePositiveRate\": 0.13793103,\n", + " \"truePositiveCount\": \"3557\",\n", + " \"falsePositiveCount\": \"76\",\n", + " \"falseNegativeCount\": \"531\",\n", + " \"trueNegativeCount\": \"475\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.85593396,\n", + " \"recall\": 0.8600783,\n", + " \"precision\": 0.98102677,\n", + " \"f1Score\": 0.9165798,\n", + " \"falsePositiveRate\": 0.123411976,\n", + " \"truePositiveCount\": \"3516\",\n", + " \"falsePositiveCount\": \"68\",\n", + " \"falseNegativeCount\": \"572\",\n", + " \"trueNegativeCount\": \"483\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.86959773,\n", + " \"recall\": 0.8500489,\n", + " \"precision\": 0.9810841,\n", + " \"f1Score\": 0.9108781,\n", + " \"falsePositiveRate\": 0.1215971,\n", + " \"truePositiveCount\": \"3475\",\n", + " \"falsePositiveCount\": \"67\",\n", + " \"falseNegativeCount\": \"613\",\n", + " \"trueNegativeCount\": \"484\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.88389677,\n", + " \"recall\": 0.83977497,\n", + " \"precision\": 0.98338586,\n", + " \"f1Score\": 0.90592426,\n", + " \"falsePositiveRate\": 0.10526316,\n", + " \"truePositiveCount\": \"3433\",\n", + " \"falsePositiveCount\": \"58\",\n", + " \"falseNegativeCount\": \"655\",\n", + " \"trueNegativeCount\": \"493\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.89513993,\n", + " \"recall\": 0.8299902,\n", + " \"precision\": 0.9840487,\n", + " \"f1Score\": 0.9004777,\n", + " \"falsePositiveRate\": 0.09981851,\n", + " \"truePositiveCount\": \"3393\",\n", + " \"falsePositiveCount\": \"55\",\n", + " \"falseNegativeCount\": \"695\",\n", + " \"trueNegativeCount\": \"496\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.90273213,\n", + " \"recall\": 0.82020545,\n", + " \"precision\": 0.9853071,\n", + " \"f1Score\": 0.8952076,\n", + " \"falsePositiveRate\": 0.0907441,\n", + " \"truePositiveCount\": \"3353\",\n", + " \"falsePositiveCount\": \"50\",\n", + " \"falseNegativeCount\": \"735\",\n", + " \"trueNegativeCount\": \"501\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9118437,\n", + " \"recall\": 0.81017613,\n", + " \"precision\": 0.98718333,\n", + " \"f1Score\": 0.88996375,\n", + " \"falsePositiveRate\": 0.07803993,\n", + " \"truePositiveCount\": \"3312\",\n", + " \"falsePositiveCount\": \"43\",\n", + " \"falseNegativeCount\": \"776\",\n", + " \"trueNegativeCount\": \"508\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9204901,\n", + " \"recall\": 0.79990214,\n", + " \"precision\": 0.98821396,\n", + " \"f1Score\": 0.8841422,\n", + " \"falsePositiveRate\": 0.0707804,\n", + " \"truePositiveCount\": \"3270\",\n", + " \"falsePositiveCount\": \"39\",\n", + " \"falseNegativeCount\": \"818\",\n", + " \"trueNegativeCount\": \"512\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9283992,\n", + " \"recall\": 0.7898728,\n", + " \"precision\": 0.9898835,\n", + " \"f1Score\": 0.87863946,\n", + " \"falsePositiveRate\": 0.05989111,\n", + " \"truePositiveCount\": \"3229\",\n", + " \"falsePositiveCount\": \"33\",\n", + " \"falseNegativeCount\": \"859\",\n", + " \"trueNegativeCount\": \"518\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.934707,\n", + " \"recall\": 0.77984345,\n", + " \"precision\": 0.9909854,\n", + " \"f1Score\": 0.8728268,\n", + " \"falsePositiveRate\": 0.05263158,\n", + " \"truePositiveCount\": \"3188\",\n", + " \"falsePositiveCount\": \"29\",\n", + " \"falseNegativeCount\": \"900\",\n", + " \"trueNegativeCount\": \"522\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9401358,\n", + " \"recall\": 0.7698141,\n", + " \"precision\": 0.99211854,\n", + " \"f1Score\": 0.86694217,\n", + " \"falsePositiveRate\": 0.04537205,\n", + " \"truePositiveCount\": \"3147\",\n", + " \"falsePositiveCount\": \"25\",\n", + " \"falseNegativeCount\": \"941\",\n", + " \"trueNegativeCount\": \"526\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.943966,\n", + " \"recall\": 0.7600294,\n", + " \"precision\": 0.9923347,\n", + " \"f1Score\": 0.86078405,\n", + " \"falsePositiveRate\": 0.043557167,\n", + " \"truePositiveCount\": \"3107\",\n", + " \"falsePositiveCount\": \"24\",\n", + " \"falseNegativeCount\": \"981\",\n", + " \"trueNegativeCount\": \"527\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9484665,\n", + " \"recall\": 0.7497554,\n", + " \"precision\": 0.9925518,\n", + " \"f1Score\": 0.85423636,\n", + " \"falsePositiveRate\": 0.041742288,\n", + " \"truePositiveCount\": \"3065\",\n", + " \"falsePositiveCount\": \"23\",\n", + " \"falseNegativeCount\": \"1023\",\n", + " \"trueNegativeCount\": \"528\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9512129,\n", + " \"recall\": 0.7399706,\n", + " \"precision\": 0.9931057,\n", + " \"f1Score\": 0.8480516,\n", + " \"falsePositiveRate\": 0.03811252,\n", + " \"truePositiveCount\": \"3025\",\n", + " \"falsePositiveCount\": \"21\",\n", + " \"falseNegativeCount\": \"1063\",\n", + " \"trueNegativeCount\": \"530\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.95533496,\n", + " \"recall\": 0.7299413,\n", + " \"precision\": 0.9946667,\n", + " \"f1Score\": 0.8419865,\n", + " \"falsePositiveRate\": 0.029038113,\n", + " \"truePositiveCount\": \"2984\",\n", + " \"falsePositiveCount\": \"16\",\n", + " \"falseNegativeCount\": \"1104\",\n", + " \"trueNegativeCount\": \"535\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.95724624,\n", + " \"recall\": 0.72015655,\n", + " \"precision\": 0.9956037,\n", + " \"f1Score\": 0.8357701,\n", + " \"falsePositiveRate\": 0.023593467,\n", + " \"truePositiveCount\": \"2944\",\n", + " \"falsePositiveCount\": \"13\",\n", + " \"falseNegativeCount\": \"1144\",\n", + " \"trueNegativeCount\": \"538\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.96008635,\n", + " \"recall\": 0.7101272,\n", + " \"precision\": 0.99588335,\n", + " \"f1Score\": 0.82907325,\n", + " \"falsePositiveRate\": 0.021778584,\n", + " \"truePositiveCount\": \"2903\",\n", + " \"falsePositiveCount\": \"12\",\n", + " \"falseNegativeCount\": \"1185\",\n", + " \"trueNegativeCount\": \"539\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9629221,\n", + " \"recall\": 0.70009786,\n", + " \"precision\": 0.99617124,\n", + " \"f1Score\": 0.82229567,\n", + " \"falsePositiveRate\": 0.019963702,\n", + " \"truePositiveCount\": \"2862\",\n", + " \"falsePositiveCount\": \"11\",\n", + " \"falseNegativeCount\": \"1226\",\n", + " \"trueNegativeCount\": \"540\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9646139,\n", + " \"recall\": 0.6900685,\n", + " \"precision\": 0.9961158,\n", + " \"f1Score\": 0.8153179,\n", + " \"falsePositiveRate\": 0.019963702,\n", + " \"truePositiveCount\": \"2821\",\n", + " \"falsePositiveCount\": \"11\",\n", + " \"falseNegativeCount\": \"1267\",\n", + " \"trueNegativeCount\": \"540\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.96656674,\n", + " \"recall\": 0.67979455,\n", + " \"precision\": 0.99605733,\n", + " \"f1Score\": 0.8080838,\n", + " \"falsePositiveRate\": 0.019963702,\n", + " \"truePositiveCount\": \"2779\",\n", + " \"falsePositiveCount\": \"11\",\n", + " \"falseNegativeCount\": \"1309\",\n", + " \"trueNegativeCount\": \"540\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.96897614,\n", + " \"recall\": 0.6700098,\n", + " \"precision\": 0.9963623,\n", + " \"f1Score\": 0.8012286,\n", + " \"falsePositiveRate\": 0.01814882,\n", + " \"truePositiveCount\": \"2739\",\n", + " \"falsePositiveCount\": \"10\",\n", + " \"falseNegativeCount\": \"1349\",\n", + " \"trueNegativeCount\": \"541\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9706901,\n", + " \"recall\": 0.6599804,\n", + " \"precision\": 0.99630725,\n", + " \"f1Score\": 0.79399645,\n", + " \"falsePositiveRate\": 0.01814882,\n", + " \"truePositiveCount\": \"2698\",\n", + " \"falsePositiveCount\": \"10\",\n", + " \"falseNegativeCount\": \"1390\",\n", + " \"trueNegativeCount\": \"541\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.97229564,\n", + " \"recall\": 0.6499511,\n", + " \"precision\": 0.99699813,\n", + " \"f1Score\": 0.7869095,\n", + " \"falsePositiveRate\": 0.014519056,\n", + " \"truePositiveCount\": \"2657\",\n", + " \"falsePositiveCount\": \"8\",\n", + " \"falseNegativeCount\": \"1431\",\n", + " \"trueNegativeCount\": \"543\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9738624,\n", + " \"recall\": 0.6399217,\n", + " \"precision\": 0.9969512,\n", + " \"f1Score\": 0.7794994,\n", + " \"falsePositiveRate\": 0.014519056,\n", + " \"truePositiveCount\": \"2616\",\n", + " \"falsePositiveCount\": \"8\",\n", + " \"falseNegativeCount\": \"1472\",\n", + " \"trueNegativeCount\": \"543\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.97582275,\n", + " \"recall\": 0.62989235,\n", + " \"precision\": 0.99728894,\n", + " \"f1Score\": 0.7721139,\n", + " \"falsePositiveRate\": 0.012704174,\n", + " \"truePositiveCount\": \"2575\",\n", + " \"falsePositiveCount\": \"7\",\n", + " \"falseNegativeCount\": \"1513\",\n", + " \"trueNegativeCount\": \"544\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9772092,\n", + " \"recall\": 0.62010765,\n", + " \"precision\": 0.9976387,\n", + " \"f1Score\": 0.76482123,\n", + " \"falsePositiveRate\": 0.010889292,\n", + " \"truePositiveCount\": \"2535\",\n", + " \"falsePositiveCount\": \"6\",\n", + " \"falseNegativeCount\": \"1553\",\n", + " \"trueNegativeCount\": \"545\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9781838,\n", + " \"recall\": 0.6100783,\n", + " \"precision\": 0.9979992,\n", + " \"f1Score\": 0.7572491,\n", + " \"falsePositiveRate\": 0.00907441,\n", + " \"truePositiveCount\": \"2494\",\n", + " \"falsePositiveCount\": \"5\",\n", + " \"falseNegativeCount\": \"1594\",\n", + " \"trueNegativeCount\": \"546\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9794321,\n", + " \"recall\": 0.5998043,\n", + " \"precision\": 0.99837136,\n", + " \"f1Score\": 0.74938875,\n", + " \"falsePositiveRate\": 0.007259528,\n", + " \"truePositiveCount\": \"2452\",\n", + " \"falsePositiveCount\": \"4\",\n", + " \"falseNegativeCount\": \"1636\",\n", + " \"trueNegativeCount\": \"547\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9804444,\n", + " \"recall\": 0.5900196,\n", + " \"precision\": 0.99834436,\n", + " \"f1Score\": 0.74169743,\n", + " \"falsePositiveRate\": 0.007259528,\n", + " \"truePositiveCount\": \"2412\",\n", + " \"falsePositiveCount\": \"4\",\n", + " \"falseNegativeCount\": \"1676\",\n", + " \"trueNegativeCount\": \"547\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9813577,\n", + " \"recall\": 0.5799902,\n", + " \"precision\": 0.9991572,\n", + " \"f1Score\": 0.7339421,\n", + " \"falsePositiveRate\": 0.003629764,\n", + " \"truePositiveCount\": \"2371\",\n", + " \"falsePositiveCount\": \"2\",\n", + " \"falseNegativeCount\": \"1717\",\n", + " \"trueNegativeCount\": \"549\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9822546,\n", + " \"recall\": 0.56996083,\n", + " \"precision\": 0.99914235,\n", + " \"f1Score\": 0.7258567,\n", + " \"falsePositiveRate\": 0.003629764,\n", + " \"truePositiveCount\": \"2330\",\n", + " \"falsePositiveCount\": \"2\",\n", + " \"falseNegativeCount\": \"1758\",\n", + " \"trueNegativeCount\": \"549\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9832565,\n", + " \"recall\": 0.5599315,\n", + " \"precision\": 0.99912703,\n", + " \"f1Score\": 0.71766734,\n", + " \"falsePositiveRate\": 0.003629764,\n", + " \"truePositiveCount\": \"2289\",\n", + " \"falsePositiveCount\": \"2\",\n", + " \"falseNegativeCount\": \"1799\",\n", + " \"trueNegativeCount\": \"549\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.983971,\n", + " \"recall\": 0.54990214,\n", + " \"precision\": 0.99955535,\n", + " \"f1Score\": 0.709484,\n", + " \"falsePositiveRate\": 0.001814882,\n", + " \"truePositiveCount\": \"2248\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"1840\",\n", + " \"trueNegativeCount\": \"550\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98478603,\n", + " \"recall\": 0.54011744,\n", + " \"precision\": 0.9995473,\n", + " \"f1Score\": 0.7012863,\n", + " \"falsePositiveRate\": 0.001814882,\n", + " \"truePositiveCount\": \"2208\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"1880\",\n", + " \"trueNegativeCount\": \"550\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98576885,\n", + " \"recall\": 0.52984345,\n", + " \"precision\": 0.99953854,\n", + " \"f1Score\": 0.6925659,\n", + " \"falsePositiveRate\": 0.001814882,\n", + " \"truePositiveCount\": \"2166\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"1922\",\n", + " \"trueNegativeCount\": \"550\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9865271,\n", + " \"recall\": 0.5198141,\n", + " \"precision\": 0.99952966,\n", + " \"f1Score\": 0.6839395,\n", + " \"falsePositiveRate\": 0.001814882,\n", + " \"truePositiveCount\": \"2125\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"1963\",\n", + " \"trueNegativeCount\": \"550\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98722655,\n", + " \"recall\": 0.5100294,\n", + " \"precision\": 0.9995206,\n", + " \"f1Score\": 0.675413,\n", + " \"falsePositiveRate\": 0.001814882,\n", + " \"truePositiveCount\": \"2085\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2003\",\n", + " \"trueNegativeCount\": \"550\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98776835,\n", + " \"recall\": 0.49975538,\n", + " \"precision\": 0.99951077,\n", + " \"f1Score\": 0.66634053,\n", + " \"falsePositiveRate\": 0.001814882,\n", + " \"truePositiveCount\": \"2043\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2045\",\n", + " \"trueNegativeCount\": \"550\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98843795,\n", + " \"recall\": 0.49021527,\n", + " \"precision\": 0.9995012,\n", + " \"f1Score\": 0.657804,\n", + " \"falsePositiveRate\": 0.001814882,\n", + " \"truePositiveCount\": \"2004\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2084\",\n", + " \"trueNegativeCount\": \"550\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98898214,\n", + " \"recall\": 0.4801859,\n", + " \"precision\": 0.99949086,\n", + " \"f1Score\": 0.64871114,\n", + " \"falsePositiveRate\": 0.001814882,\n", + " \"truePositiveCount\": \"1963\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2125\",\n", + " \"trueNegativeCount\": \"550\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98952276,\n", + " \"recall\": 0.46991193,\n", + " \"precision\": 0.9994797,\n", + " \"f1Score\": 0.63926786,\n", + " \"falsePositiveRate\": 0.001814882,\n", + " \"truePositiveCount\": \"1921\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2167\",\n", + " \"trueNegativeCount\": \"550\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99007314,\n", + " \"recall\": 0.4598826,\n", + " \"precision\": 0.9994684,\n", + " \"f1Score\": 0.62992126,\n", + " \"falsePositiveRate\": 0.001814882,\n", + " \"truePositiveCount\": \"1880\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2208\",\n", + " \"trueNegativeCount\": \"550\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.990457,\n", + " \"recall\": 0.45009786,\n", + " \"precision\": 0.9994568,\n", + " \"f1Score\": 0.620678,\n", + " \"falsePositiveRate\": 0.001814882,\n", + " \"truePositiveCount\": \"1840\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2248\",\n", + " \"trueNegativeCount\": \"550\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99079466,\n", + " \"recall\": 0.44006848,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.61117715,\n", + " \"truePositiveCount\": \"1799\",\n", + " \"falseNegativeCount\": \"2289\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9911339,\n", + " \"recall\": 0.43003914,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.60143685,\n", + " \"truePositiveCount\": \"1758\",\n", + " \"falseNegativeCount\": \"2330\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99141955,\n", + " \"recall\": 0.4200098,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.591559,\n", + " \"truePositiveCount\": \"1717\",\n", + " \"falseNegativeCount\": \"2371\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9917892,\n", + " \"recall\": 0.40998042,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.5815406,\n", + " \"truePositiveCount\": \"1676\",\n", + " \"falseNegativeCount\": \"2412\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9921363,\n", + " \"recall\": 0.39995107,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.57137865,\n", + " \"truePositiveCount\": \"1635\",\n", + " \"falseNegativeCount\": \"2453\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99253094,\n", + " \"recall\": 0.38992172,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.56107,\n", + " \"truePositiveCount\": \"1594\",\n", + " \"falseNegativeCount\": \"2494\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9928577,\n", + " \"recall\": 0.380137,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.5508685,\n", + " \"truePositiveCount\": \"1554\",\n", + " \"falseNegativeCount\": \"2534\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99313205,\n", + " \"recall\": 0.369863,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.54,\n", + " \"truePositiveCount\": \"1512\",\n", + " \"falseNegativeCount\": \"2576\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9933454,\n", + " \"recall\": 0.35983366,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.5292319,\n", + " \"truePositiveCount\": \"1471\",\n", + " \"falseNegativeCount\": \"2617\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99355274,\n", + " \"recall\": 0.35004893,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.5185722,\n", + " \"truePositiveCount\": \"1431\",\n", + " \"falseNegativeCount\": \"2657\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99378794,\n", + " \"recall\": 0.33977494,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.507212,\n", + " \"truePositiveCount\": \"1389\",\n", + " \"falseNegativeCount\": \"2699\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9939992,\n", + " \"recall\": 0.33023483,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.49650607,\n", + " \"truePositiveCount\": \"1350\",\n", + " \"falseNegativeCount\": \"2738\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9942213,\n", + " \"recall\": 0.31996086,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.48480356,\n", + " \"truePositiveCount\": \"1308\",\n", + " \"falseNegativeCount\": \"2780\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9944552,\n", + " \"recall\": 0.30993152,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.47320262,\n", + " \"truePositiveCount\": \"1267\",\n", + " \"falseNegativeCount\": \"2821\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9946339,\n", + " \"recall\": 0.29990214,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.46142265,\n", + " \"truePositiveCount\": \"1226\",\n", + " \"falseNegativeCount\": \"2862\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99486357,\n", + " \"recall\": 0.2898728,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.44945952,\n", + " \"truePositiveCount\": \"1185\",\n", + " \"falseNegativeCount\": \"2903\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9950511,\n", + " \"recall\": 0.28008807,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.4376075,\n", + " \"truePositiveCount\": \"1145\",\n", + " \"falseNegativeCount\": \"2943\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9953063,\n", + " \"recall\": 0.2698141,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.42496628,\n", + " \"truePositiveCount\": \"1103\",\n", + " \"falseNegativeCount\": \"2985\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99548787,\n", + " \"recall\": 0.26002935,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.4127354,\n", + " \"truePositiveCount\": \"1063\",\n", + " \"falseNegativeCount\": \"3025\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99564606,\n", + " \"recall\": 0.25024462,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.40031305,\n", + " \"truePositiveCount\": \"1023\",\n", + " \"falseNegativeCount\": \"3065\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9958494,\n", + " \"recall\": 0.23997064,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.3870586,\n", + " \"truePositiveCount\": \"981\",\n", + " \"falseNegativeCount\": \"3107\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9959897,\n", + " \"recall\": 0.2299413,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.37390614,\n", + " \"truePositiveCount\": \"940\",\n", + " \"falseNegativeCount\": \"3148\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9961461,\n", + " \"recall\": 0.21991193,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.3605374,\n", + " \"truePositiveCount\": \"899\",\n", + " \"falseNegativeCount\": \"3189\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99626255,\n", + " \"recall\": 0.20988259,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.346947,\n", + " \"truePositiveCount\": \"858\",\n", + " \"falseNegativeCount\": \"3230\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99635655,\n", + " \"recall\": 0.19985323,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.33312947,\n", + " \"truePositiveCount\": \"817\",\n", + " \"falseNegativeCount\": \"3271\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9964815,\n", + " \"recall\": 0.1900685,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.31942445,\n", + " \"truePositiveCount\": \"777\",\n", + " \"falseNegativeCount\": \"3311\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9966108,\n", + " \"recall\": 0.17979452,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.30478954,\n", + " \"truePositiveCount\": \"735\",\n", + " \"falseNegativeCount\": \"3353\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99673027,\n", + " \"recall\": 0.16976516,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.29025513,\n", + " \"truePositiveCount\": \"694\",\n", + " \"falseNegativeCount\": \"3394\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99681914,\n", + " \"recall\": 0.15998043,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.27583298,\n", + " \"truePositiveCount\": \"654\",\n", + " \"falseNegativeCount\": \"3434\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9969455,\n", + " \"recall\": 0.15019569,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.26116547,\n", + " \"truePositiveCount\": \"614\",\n", + " \"falseNegativeCount\": \"3474\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99705213,\n", + " \"recall\": 0.13992172,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.24549356,\n", + " \"truePositiveCount\": \"572\",\n", + " \"falseNegativeCount\": \"3516\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9971576,\n", + " \"recall\": 0.13013698,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.23030303,\n", + " \"truePositiveCount\": \"532\",\n", + " \"falseNegativeCount\": \"3556\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9972461,\n", + " \"recall\": 0.11986301,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.21406728,\n", + " \"truePositiveCount\": \"490\",\n", + " \"falseNegativeCount\": \"3598\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9973425,\n", + " \"recall\": 0.110078275,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.19832525,\n", + " \"truePositiveCount\": \"450\",\n", + " \"falseNegativeCount\": \"3638\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9974355,\n", + " \"recall\": 0.10004892,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.18189904,\n", + " \"truePositiveCount\": \"409\",\n", + " \"falseNegativeCount\": \"3679\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9975045,\n", + " \"recall\": 0.09001957,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.16517055,\n", + " \"truePositiveCount\": \"368\",\n", + " \"falseNegativeCount\": \"3720\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99760437,\n", + " \"recall\": 0.079990216,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.14813137,\n", + " \"truePositiveCount\": \"327\",\n", + " \"falseNegativeCount\": \"3761\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9976864,\n", + " \"recall\": 0.07020548,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.1312,\n", + " \"truePositiveCount\": \"287\",\n", + " \"falseNegativeCount\": \"3801\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9977548,\n", + " \"recall\": 0.059931505,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.11308562,\n", + " \"truePositiveCount\": \"245\",\n", + " \"falseNegativeCount\": \"3843\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99784184,\n", + " \"recall\": 0.05014677,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.095504306,\n", + " \"truePositiveCount\": \"205\",\n", + " \"falseNegativeCount\": \"3883\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9979226,\n", + " \"recall\": 0.040117417,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.07714017,\n", + " \"truePositiveCount\": \"164\",\n", + " \"falseNegativeCount\": \"3924\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9980229,\n", + " \"recall\": 0.030088063,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.058418427,\n", + " \"truePositiveCount\": \"123\",\n", + " \"falseNegativeCount\": \"3965\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9981275,\n", + " \"recall\": 0.01981409,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.03885824,\n", + " \"truePositiveCount\": \"81\",\n", + " \"falseNegativeCount\": \"4007\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9982374,\n", + " \"recall\": 0.009784736,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.019379845,\n", + " \"truePositiveCount\": \"40\",\n", + " \"falseNegativeCount\": \"4048\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9987167,\n", + " \"precision\": \"NaN\",\n", + " \"f1Score\": \"NaN\",\n", + " \"falseNegativeCount\": \"4088\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 1.0,\n", + " \"precision\": \"NaN\",\n", + " \"f1Score\": \"NaN\",\n", + " \"falseNegativeCount\": \"4088\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 2147483647\n", + " }\n", + " ],\n", + " \"auRoc\": 0.9340445,\n", + " \"logLoss\": 0.09388557\n", + " },\n", + " \"displayName\": \"1\"\n", + " },\n", + " {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/TBL8310014324731019264/modelEvaluations/1660852072267559125\",\n", + " \"annotationSpecId\": \"not available\",\n", + " \"createTime\": \"2021-02-28T17:24:31.800758Z\",\n", + " \"evaluatedExampleCount\": 551,\n", + " \"classificationEvaluationMetrics\": {\n", + " \"auPrc\": 0.61527693,\n", + " \"confidenceMetricsEntry\": [\n", + " {\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.1187756,\n", + " \"f1Score\": 0.2123314,\n", + " \"falsePositiveRate\": 1.0,\n", + " \"truePositiveCount\": \"551\",\n", + " \"falsePositiveCount\": \"4088\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.009267787,\n", + " \"recall\": 0.9981851,\n", + " \"precision\": 0.19400352,\n", + " \"f1Score\": 0.3248671,\n", + " \"falsePositiveRate\": 0.55895305,\n", + " \"truePositiveCount\": \"550\",\n", + " \"falsePositiveCount\": \"2285\",\n", + " \"falseNegativeCount\": \"1\",\n", + " \"trueNegativeCount\": \"1803\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.022539802,\n", + " \"recall\": 0.9891107,\n", + " \"precision\": 0.2587844,\n", + " \"f1Score\": 0.4102371,\n", + " \"falsePositiveRate\": 0.38184932,\n", + " \"truePositiveCount\": \"545\",\n", + " \"falsePositiveCount\": \"1561\",\n", + " \"falseNegativeCount\": \"6\",\n", + " \"trueNegativeCount\": \"2527\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.03954769,\n", + " \"recall\": 0.9782214,\n", + " \"precision\": 0.3113807,\n", + " \"f1Score\": 0.47239265,\n", + " \"falsePositiveRate\": 0.29158512,\n", + " \"truePositiveCount\": \"539\",\n", + " \"falsePositiveCount\": \"1192\",\n", + " \"falseNegativeCount\": \"12\",\n", + " \"trueNegativeCount\": \"2896\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.045538254,\n", + " \"recall\": 0.969147,\n", + " \"precision\": 0.32902032,\n", + " \"f1Score\": 0.49126035,\n", + " \"falsePositiveRate\": 0.26638943,\n", + " \"truePositiveCount\": \"534\",\n", + " \"falsePositiveCount\": \"1089\",\n", + " \"falseNegativeCount\": \"17\",\n", + " \"trueNegativeCount\": \"2999\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.050852664,\n", + " \"recall\": 0.95825773,\n", + " \"precision\": 0.33802816,\n", + " \"f1Score\": 0.49976337,\n", + " \"falsePositiveRate\": 0.2529354,\n", + " \"truePositiveCount\": \"528\",\n", + " \"falsePositiveCount\": \"1034\",\n", + " \"falseNegativeCount\": \"23\",\n", + " \"trueNegativeCount\": \"3054\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.063637815,\n", + " \"recall\": 0.9491833,\n", + " \"precision\": 0.3647141,\n", + " \"f1Score\": 0.52695215,\n", + " \"falsePositiveRate\": 0.22284736,\n", + " \"truePositiveCount\": \"523\",\n", + " \"falsePositiveCount\": \"911\",\n", + " \"falseNegativeCount\": \"28\",\n", + " \"trueNegativeCount\": \"3177\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.07210142,\n", + " \"recall\": 0.938294,\n", + " \"precision\": 0.377097,\n", + " \"f1Score\": 0.5379813,\n", + " \"falsePositiveRate\": 0.2089041,\n", + " \"truePositiveCount\": \"517\",\n", + " \"falsePositiveCount\": \"854\",\n", + " \"falseNegativeCount\": \"34\",\n", + " \"trueNegativeCount\": \"3234\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.07750216,\n", + " \"recall\": 0.9292196,\n", + " \"precision\": 0.3826607,\n", + " \"f1Score\": 0.54208577,\n", + " \"falsePositiveRate\": 0.2020548,\n", + " \"truePositiveCount\": \"512\",\n", + " \"falsePositiveCount\": \"826\",\n", + " \"falseNegativeCount\": \"39\",\n", + " \"trueNegativeCount\": \"3262\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.09185968,\n", + " \"recall\": 0.9183303,\n", + " \"precision\": 0.4,\n", + " \"f1Score\": 0.55726874,\n", + " \"falsePositiveRate\": 0.18566537,\n", + " \"truePositiveCount\": \"506\",\n", + " \"falsePositiveCount\": \"759\",\n", + " \"falseNegativeCount\": \"45\",\n", + " \"trueNegativeCount\": \"3329\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.09674809,\n", + " \"recall\": 0.9092559,\n", + " \"precision\": 0.40501213,\n", + " \"f1Score\": 0.5604027,\n", + " \"falsePositiveRate\": 0.18003914,\n", + " \"truePositiveCount\": \"501\",\n", + " \"falsePositiveCount\": \"736\",\n", + " \"falseNegativeCount\": \"50\",\n", + " \"trueNegativeCount\": \"3352\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.10816234,\n", + " \"recall\": 0.89836663,\n", + " \"precision\": 0.4212766,\n", + " \"f1Score\": 0.5735805,\n", + " \"falsePositiveRate\": 0.16634052,\n", + " \"truePositiveCount\": \"495\",\n", + " \"falsePositiveCount\": \"680\",\n", + " \"falseNegativeCount\": \"56\",\n", + " \"trueNegativeCount\": \"3408\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.12236822,\n", + " \"recall\": 0.8892922,\n", + " \"precision\": 0.43672013,\n", + " \"f1Score\": 0.58577406,\n", + " \"falsePositiveRate\": 0.15459883,\n", + " \"truePositiveCount\": \"490\",\n", + " \"falsePositiveCount\": \"632\",\n", + " \"falseNegativeCount\": \"61\",\n", + " \"trueNegativeCount\": \"3456\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.1267207,\n", + " \"recall\": 0.8784029,\n", + " \"precision\": 0.4368231,\n", + " \"f1Score\": 0.58348405,\n", + " \"falsePositiveRate\": 0.15264188,\n", + " \"truePositiveCount\": \"484\",\n", + " \"falsePositiveCount\": \"624\",\n", + " \"falseNegativeCount\": \"67\",\n", + " \"trueNegativeCount\": \"3464\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.15273023,\n", + " \"recall\": 0.8693285,\n", + " \"precision\": 0.46550047,\n", + " \"f1Score\": 0.60632914,\n", + " \"falsePositiveRate\": 0.13454011,\n", + " \"truePositiveCount\": \"479\",\n", + " \"falsePositiveCount\": \"550\",\n", + " \"falseNegativeCount\": \"72\",\n", + " \"trueNegativeCount\": \"3538\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.1649416,\n", + " \"recall\": 0.8584392,\n", + " \"precision\": 0.47252747,\n", + " \"f1Score\": 0.6095361,\n", + " \"falsePositiveRate\": 0.12915851,\n", + " \"truePositiveCount\": \"473\",\n", + " \"falsePositiveCount\": \"528\",\n", + " \"falseNegativeCount\": \"78\",\n", + " \"trueNegativeCount\": \"3560\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.18867286,\n", + " \"recall\": 0.8493648,\n", + " \"precision\": 0.49628845,\n", + " \"f1Score\": 0.62650603,\n", + " \"falsePositiveRate\": 0.116193734,\n", + " \"truePositiveCount\": \"468\",\n", + " \"falsePositiveCount\": \"475\",\n", + " \"falseNegativeCount\": \"83\",\n", + " \"trueNegativeCount\": \"3613\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.19824302,\n", + " \"recall\": 0.8384755,\n", + " \"precision\": 0.5038168,\n", + " \"f1Score\": 0.6294278,\n", + " \"falsePositiveRate\": 0.11130137,\n", + " \"truePositiveCount\": \"462\",\n", + " \"falsePositiveCount\": \"455\",\n", + " \"falseNegativeCount\": \"89\",\n", + " \"trueNegativeCount\": \"3633\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.20855853,\n", + " \"recall\": 0.8294011,\n", + " \"precision\": 0.5163842,\n", + " \"f1Score\": 0.6364902,\n", + " \"falsePositiveRate\": 0.104696676,\n", + " \"truePositiveCount\": \"457\",\n", + " \"falsePositiveCount\": \"428\",\n", + " \"falseNegativeCount\": \"94\",\n", + " \"trueNegativeCount\": \"3660\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.2164171,\n", + " \"recall\": 0.8185118,\n", + " \"precision\": 0.51898736,\n", + " \"f1Score\": 0.6352113,\n", + " \"falsePositiveRate\": 0.10225049,\n", + " \"truePositiveCount\": \"451\",\n", + " \"falsePositiveCount\": \"418\",\n", + " \"falseNegativeCount\": \"100\",\n", + " \"trueNegativeCount\": \"3670\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.2228882,\n", + " \"recall\": 0.8094374,\n", + " \"precision\": 0.52347416,\n", + " \"f1Score\": 0.63578045,\n", + " \"falsePositiveRate\": 0.09931507,\n", + " \"truePositiveCount\": \"446\",\n", + " \"falsePositiveCount\": \"406\",\n", + " \"falseNegativeCount\": \"105\",\n", + " \"trueNegativeCount\": \"3682\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.2343938,\n", + " \"recall\": 0.7985481,\n", + " \"precision\": 0.5339806,\n", + " \"f1Score\": 0.64,\n", + " \"falsePositiveRate\": 0.09393346,\n", + " \"truePositiveCount\": \"440\",\n", + " \"falsePositiveCount\": \"384\",\n", + " \"falseNegativeCount\": \"111\",\n", + " \"trueNegativeCount\": \"3704\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.24437484,\n", + " \"recall\": 0.7894737,\n", + " \"precision\": 0.5390335,\n", + " \"f1Score\": 0.640648,\n", + " \"falsePositiveRate\": 0.090998046,\n", + " \"truePositiveCount\": \"435\",\n", + " \"falsePositiveCount\": \"372\",\n", + " \"falseNegativeCount\": \"116\",\n", + " \"trueNegativeCount\": \"3716\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.24899487,\n", + " \"recall\": 0.7785844,\n", + " \"precision\": 0.53692114,\n", + " \"f1Score\": 0.63555557,\n", + " \"falsePositiveRate\": 0.0905088,\n", + " \"truePositiveCount\": \"429\",\n", + " \"falsePositiveCount\": \"370\",\n", + " \"falseNegativeCount\": \"122\",\n", + " \"trueNegativeCount\": \"3718\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.25527748,\n", + " \"recall\": 0.76951,\n", + " \"precision\": 0.5401274,\n", + " \"f1Score\": 0.6347305,\n", + " \"falsePositiveRate\": 0.08830724,\n", + " \"truePositiveCount\": \"424\",\n", + " \"falsePositiveCount\": \"361\",\n", + " \"falseNegativeCount\": \"127\",\n", + " \"trueNegativeCount\": \"3727\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.2643137,\n", + " \"recall\": 0.7586207,\n", + " \"precision\": 0.5414508,\n", + " \"f1Score\": 0.6318972,\n", + " \"falsePositiveRate\": 0.08659491,\n", + " \"truePositiveCount\": \"418\",\n", + " \"falsePositiveCount\": \"354\",\n", + " \"falseNegativeCount\": \"133\",\n", + " \"trueNegativeCount\": \"3734\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.27015588,\n", + " \"recall\": 0.7495463,\n", + " \"precision\": 0.544137,\n", + " \"f1Score\": 0.63053435,\n", + " \"falsePositiveRate\": 0.08463796,\n", + " \"truePositiveCount\": \"413\",\n", + " \"falsePositiveCount\": \"346\",\n", + " \"falseNegativeCount\": \"138\",\n", + " \"trueNegativeCount\": \"3742\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.2785622,\n", + " \"recall\": 0.738657,\n", + " \"precision\": 0.547043,\n", + " \"f1Score\": 0.62857145,\n", + " \"falsePositiveRate\": 0.0824364,\n", + " \"truePositiveCount\": \"407\",\n", + " \"falsePositiveCount\": \"337\",\n", + " \"falseNegativeCount\": \"144\",\n", + " \"trueNegativeCount\": \"3751\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.29147997,\n", + " \"recall\": 0.72958255,\n", + " \"precision\": 0.5552486,\n", + " \"f1Score\": 0.63058823,\n", + " \"falsePositiveRate\": 0.07876712,\n", + " \"truePositiveCount\": \"402\",\n", + " \"falsePositiveCount\": \"322\",\n", + " \"falseNegativeCount\": \"149\",\n", + " \"trueNegativeCount\": \"3766\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.30125153,\n", + " \"recall\": 0.71869326,\n", + " \"precision\": 0.55774647,\n", + " \"f1Score\": 0.628073,\n", + " \"falsePositiveRate\": 0.07681017,\n", + " \"truePositiveCount\": \"396\",\n", + " \"falsePositiveCount\": \"314\",\n", + " \"falseNegativeCount\": \"155\",\n", + " \"trueNegativeCount\": \"3774\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.30796623,\n", + " \"recall\": 0.70961887,\n", + " \"precision\": 0.55857146,\n", + " \"f1Score\": 0.6250999,\n", + " \"falsePositiveRate\": 0.07558709,\n", + " \"truePositiveCount\": \"391\",\n", + " \"falsePositiveCount\": \"309\",\n", + " \"falseNegativeCount\": \"160\",\n", + " \"trueNegativeCount\": \"3779\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.32498023,\n", + " \"recall\": 0.6987296,\n", + " \"precision\": 0.5636896,\n", + " \"f1Score\": 0.623987,\n", + " \"falsePositiveRate\": 0.07289628,\n", + " \"truePositiveCount\": \"385\",\n", + " \"falsePositiveCount\": \"298\",\n", + " \"falseNegativeCount\": \"166\",\n", + " \"trueNegativeCount\": \"3790\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.33223236,\n", + " \"recall\": 0.6896552,\n", + " \"precision\": 0.5654762,\n", + " \"f1Score\": 0.6214227,\n", + " \"falsePositiveRate\": 0.071428575,\n", + " \"truePositiveCount\": \"380\",\n", + " \"falsePositiveCount\": \"292\",\n", + " \"falseNegativeCount\": \"171\",\n", + " \"trueNegativeCount\": \"3796\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.3348615,\n", + " \"recall\": 0.6787659,\n", + " \"precision\": 0.562406,\n", + " \"f1Score\": 0.61513156,\n", + " \"falsePositiveRate\": 0.07118395,\n", + " \"truePositiveCount\": \"374\",\n", + " \"falsePositiveCount\": \"291\",\n", + " \"falseNegativeCount\": \"177\",\n", + " \"trueNegativeCount\": \"3797\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.33746937,\n", + " \"recall\": 0.66969144,\n", + " \"precision\": 0.5607903,\n", + " \"f1Score\": 0.61042184,\n", + " \"falsePositiveRate\": 0.070694715,\n", + " \"truePositiveCount\": \"369\",\n", + " \"falsePositiveCount\": \"289\",\n", + " \"falseNegativeCount\": \"182\",\n", + " \"trueNegativeCount\": \"3799\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.34359238,\n", + " \"recall\": 0.65880215,\n", + " \"precision\": 0.561051,\n", + " \"f1Score\": 0.60601,\n", + " \"falsePositiveRate\": 0.06947163,\n", + " \"truePositiveCount\": \"363\",\n", + " \"falsePositiveCount\": \"284\",\n", + " \"falseNegativeCount\": \"188\",\n", + " \"trueNegativeCount\": \"3804\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.3495217,\n", + " \"recall\": 0.64972776,\n", + " \"precision\": 0.5585023,\n", + " \"f1Score\": 0.6006711,\n", + " \"falsePositiveRate\": 0.069227,\n", + " \"truePositiveCount\": \"358\",\n", + " \"falsePositiveCount\": \"283\",\n", + " \"falseNegativeCount\": \"193\",\n", + " \"trueNegativeCount\": \"3805\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.35605764,\n", + " \"recall\": 0.63883847,\n", + " \"precision\": 0.5623003,\n", + " \"f1Score\": 0.5981308,\n", + " \"falsePositiveRate\": 0.06702544,\n", + " \"truePositiveCount\": \"352\",\n", + " \"falsePositiveCount\": \"274\",\n", + " \"falseNegativeCount\": \"199\",\n", + " \"trueNegativeCount\": \"3814\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.36666742,\n", + " \"recall\": 0.6297641,\n", + " \"precision\": 0.56978655,\n", + " \"f1Score\": 0.59827584,\n", + " \"falsePositiveRate\": 0.06409002,\n", + " \"truePositiveCount\": \"347\",\n", + " \"falsePositiveCount\": \"262\",\n", + " \"falseNegativeCount\": \"204\",\n", + " \"trueNegativeCount\": \"3826\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.37787727,\n", + " \"recall\": 0.6188748,\n", + " \"precision\": 0.5809199,\n", + " \"f1Score\": 0.599297,\n", + " \"falsePositiveRate\": 0.060176127,\n", + " \"truePositiveCount\": \"341\",\n", + " \"falsePositiveCount\": \"246\",\n", + " \"falseNegativeCount\": \"210\",\n", + " \"trueNegativeCount\": \"3842\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.39064476,\n", + " \"recall\": 0.60980034,\n", + " \"precision\": 0.58638746,\n", + " \"f1Score\": 0.59786475,\n", + " \"falsePositiveRate\": 0.05797456,\n", + " \"truePositiveCount\": \"336\",\n", + " \"falsePositiveCount\": \"237\",\n", + " \"falseNegativeCount\": \"215\",\n", + " \"trueNegativeCount\": \"3851\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.39657614,\n", + " \"recall\": 0.59891105,\n", + " \"precision\": 0.5871886,\n", + " \"f1Score\": 0.5929919,\n", + " \"falsePositiveRate\": 0.056751467,\n", + " \"truePositiveCount\": \"330\",\n", + " \"falsePositiveCount\": \"232\",\n", + " \"falseNegativeCount\": \"221\",\n", + " \"trueNegativeCount\": \"3856\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.39991632,\n", + " \"recall\": 0.58983666,\n", + " \"precision\": 0.5866426,\n", + " \"f1Score\": 0.5882353,\n", + " \"falsePositiveRate\": 0.05601761,\n", + " \"truePositiveCount\": \"325\",\n", + " \"falsePositiveCount\": \"229\",\n", + " \"falseNegativeCount\": \"226\",\n", + " \"trueNegativeCount\": \"3859\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.41141164,\n", + " \"recall\": 0.57894737,\n", + " \"precision\": 0.5896488,\n", + " \"f1Score\": 0.5842491,\n", + " \"falsePositiveRate\": 0.054305285,\n", + " \"truePositiveCount\": \"319\",\n", + " \"falsePositiveCount\": \"222\",\n", + " \"falseNegativeCount\": \"232\",\n", + " \"trueNegativeCount\": \"3866\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.4184669,\n", + " \"recall\": 0.569873,\n", + " \"precision\": 0.5902256,\n", + " \"f1Score\": 0.5798707,\n", + " \"falsePositiveRate\": 0.05332681,\n", + " \"truePositiveCount\": \"314\",\n", + " \"falsePositiveCount\": \"218\",\n", + " \"falseNegativeCount\": \"237\",\n", + " \"trueNegativeCount\": \"3870\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.42953506,\n", + " \"recall\": 0.5589837,\n", + " \"precision\": 0.5980582,\n", + " \"f1Score\": 0.5778612,\n", + " \"falsePositiveRate\": 0.05063601,\n", + " \"truePositiveCount\": \"308\",\n", + " \"falsePositiveCount\": \"207\",\n", + " \"falseNegativeCount\": \"243\",\n", + " \"trueNegativeCount\": \"3881\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.43759817,\n", + " \"recall\": 0.54990923,\n", + " \"precision\": 0.6023857,\n", + " \"f1Score\": 0.57495254,\n", + " \"falsePositiveRate\": 0.04892368,\n", + " \"truePositiveCount\": \"303\",\n", + " \"falsePositiveCount\": \"200\",\n", + " \"falseNegativeCount\": \"248\",\n", + " \"trueNegativeCount\": \"3888\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.44597256,\n", + " \"recall\": 0.53901994,\n", + " \"precision\": 0.60365856,\n", + " \"f1Score\": 0.569511,\n", + " \"falsePositiveRate\": 0.047700588,\n", + " \"truePositiveCount\": \"297\",\n", + " \"falsePositiveCount\": \"195\",\n", + " \"falseNegativeCount\": \"254\",\n", + " \"trueNegativeCount\": \"3893\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.44895133,\n", + " \"recall\": 0.52994555,\n", + " \"precision\": 0.6045549,\n", + " \"f1Score\": 0.5647969,\n", + " \"falsePositiveRate\": 0.046722114,\n", + " \"truePositiveCount\": \"292\",\n", + " \"falsePositiveCount\": \"191\",\n", + " \"falseNegativeCount\": \"259\",\n", + " \"trueNegativeCount\": \"3897\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.46139872,\n", + " \"recall\": 0.51905626,\n", + " \"precision\": 0.6085106,\n", + " \"f1Score\": 0.5602351,\n", + " \"falsePositiveRate\": 0.045009784,\n", + " \"truePositiveCount\": \"286\",\n", + " \"falsePositiveCount\": \"184\",\n", + " \"falseNegativeCount\": \"265\",\n", + " \"trueNegativeCount\": \"3904\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.47061405,\n", + " \"recall\": 0.5099819,\n", + " \"precision\": 0.61622804,\n", + " \"f1Score\": 0.55809337,\n", + " \"falsePositiveRate\": 0.04280822,\n", + " \"truePositiveCount\": \"281\",\n", + " \"falsePositiveCount\": \"175\",\n", + " \"falseNegativeCount\": \"270\",\n", + " \"trueNegativeCount\": \"3913\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.48159525,\n", + " \"recall\": 0.49909255,\n", + " \"precision\": 0.6207675,\n", + " \"f1Score\": 0.55331993,\n", + " \"falsePositiveRate\": 0.04109589,\n", + " \"truePositiveCount\": \"275\",\n", + " \"falsePositiveCount\": \"168\",\n", + " \"falseNegativeCount\": \"276\",\n", + " \"trueNegativeCount\": \"3920\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.48736593,\n", + " \"recall\": 0.48820326,\n", + " \"precision\": 0.6226852,\n", + " \"f1Score\": 0.54730415,\n", + " \"falsePositiveRate\": 0.0398728,\n", + " \"truePositiveCount\": \"269\",\n", + " \"falsePositiveCount\": \"163\",\n", + " \"falseNegativeCount\": \"282\",\n", + " \"trueNegativeCount\": \"3925\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.4939781,\n", + " \"recall\": 0.47912887,\n", + " \"precision\": 0.6300716,\n", + " \"f1Score\": 0.5443299,\n", + " \"falsePositiveRate\": 0.037915852,\n", + " \"truePositiveCount\": \"264\",\n", + " \"falsePositiveCount\": \"155\",\n", + " \"falseNegativeCount\": \"287\",\n", + " \"trueNegativeCount\": \"3933\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.4995014,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.635468,\n", + " \"f1Score\": 0.5391849,\n", + " \"falsePositiveRate\": 0.036203522,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"148\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3940\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5025096,\n", + " \"recall\": 0.45916516,\n", + " \"precision\": 0.63408524,\n", + " \"f1Score\": 0.5326316,\n", + " \"falsePositiveRate\": 0.035714287,\n", + " \"truePositiveCount\": \"253\",\n", + " \"falsePositiveCount\": \"146\",\n", + " \"falseNegativeCount\": \"298\",\n", + " \"trueNegativeCount\": \"3942\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.51137525,\n", + " \"recall\": 0.44827586,\n", + " \"precision\": 0.63010204,\n", + " \"f1Score\": 0.52386004,\n", + " \"falsePositiveRate\": 0.035469666,\n", + " \"truePositiveCount\": \"247\",\n", + " \"falsePositiveCount\": \"145\",\n", + " \"falseNegativeCount\": \"304\",\n", + " \"trueNegativeCount\": \"3943\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.51734245,\n", + " \"recall\": 0.43920144,\n", + " \"precision\": 0.62857145,\n", + " \"f1Score\": 0.517094,\n", + " \"falsePositiveRate\": 0.03498043,\n", + " \"truePositiveCount\": \"242\",\n", + " \"falsePositiveCount\": \"143\",\n", + " \"falseNegativeCount\": \"309\",\n", + " \"trueNegativeCount\": \"3945\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5248498,\n", + " \"recall\": 0.42831215,\n", + " \"precision\": 0.631016,\n", + " \"f1Score\": 0.5102703,\n", + " \"falsePositiveRate\": 0.03375734,\n", + " \"truePositiveCount\": \"236\",\n", + " \"falsePositiveCount\": \"138\",\n", + " \"falseNegativeCount\": \"315\",\n", + " \"trueNegativeCount\": \"3950\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.53241414,\n", + " \"recall\": 0.41923776,\n", + " \"precision\": 0.64705884,\n", + " \"f1Score\": 0.5088106,\n", + " \"falsePositiveRate\": 0.030821918,\n", + " \"truePositiveCount\": \"231\",\n", + " \"falsePositiveCount\": \"126\",\n", + " \"falseNegativeCount\": \"320\",\n", + " \"trueNegativeCount\": \"3962\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5427009,\n", + " \"recall\": 0.40834847,\n", + " \"precision\": 0.64841497,\n", + " \"f1Score\": 0.5011136,\n", + " \"falsePositiveRate\": 0.029843444,\n", + " \"truePositiveCount\": \"225\",\n", + " \"falsePositiveCount\": \"122\",\n", + " \"falseNegativeCount\": \"326\",\n", + " \"trueNegativeCount\": \"3966\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5505563,\n", + " \"recall\": 0.39927405,\n", + " \"precision\": 0.6547619,\n", + " \"f1Score\": 0.4960541,\n", + " \"falsePositiveRate\": 0.028375734,\n", + " \"truePositiveCount\": \"220\",\n", + " \"falsePositiveCount\": \"116\",\n", + " \"falseNegativeCount\": \"331\",\n", + " \"trueNegativeCount\": \"3972\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.55334836,\n", + " \"recall\": 0.38838476,\n", + " \"precision\": 0.65045595,\n", + " \"f1Score\": 0.48636365,\n", + " \"falsePositiveRate\": 0.028131116,\n", + " \"truePositiveCount\": \"214\",\n", + " \"falsePositiveCount\": \"115\",\n", + " \"falseNegativeCount\": \"337\",\n", + " \"trueNegativeCount\": \"3973\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.55837405,\n", + " \"recall\": 0.37931034,\n", + " \"precision\": 0.6490683,\n", + " \"f1Score\": 0.4788087,\n", + " \"falsePositiveRate\": 0.02764188,\n", + " \"truePositiveCount\": \"209\",\n", + " \"falsePositiveCount\": \"113\",\n", + " \"falseNegativeCount\": \"342\",\n", + " \"trueNegativeCount\": \"3975\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5685674,\n", + " \"recall\": 0.36842105,\n", + " \"precision\": 0.650641,\n", + " \"f1Score\": 0.47045192,\n", + " \"falsePositiveRate\": 0.026663406,\n", + " \"truePositiveCount\": \"203\",\n", + " \"falsePositiveCount\": \"109\",\n", + " \"falseNegativeCount\": \"348\",\n", + " \"trueNegativeCount\": \"3979\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5732897,\n", + " \"recall\": 0.35934663,\n", + " \"precision\": 0.6578073,\n", + " \"f1Score\": 0.46478873,\n", + " \"falsePositiveRate\": 0.025195695,\n", + " \"truePositiveCount\": \"198\",\n", + " \"falsePositiveCount\": \"103\",\n", + " \"falseNegativeCount\": \"353\",\n", + " \"trueNegativeCount\": \"3985\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5798386,\n", + " \"recall\": 0.34845734,\n", + " \"precision\": 0.65753424,\n", + " \"f1Score\": 0.455516,\n", + " \"falsePositiveRate\": 0.02446184,\n", + " \"truePositiveCount\": \"192\",\n", + " \"falsePositiveCount\": \"100\",\n", + " \"falseNegativeCount\": \"359\",\n", + " \"trueNegativeCount\": \"3988\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5841479,\n", + " \"recall\": 0.33938295,\n", + " \"precision\": 0.6561403,\n", + " \"f1Score\": 0.4473684,\n", + " \"falsePositiveRate\": 0.023972603,\n", + " \"truePositiveCount\": \"187\",\n", + " \"falsePositiveCount\": \"98\",\n", + " \"falseNegativeCount\": \"364\",\n", + " \"trueNegativeCount\": \"3990\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.58893955,\n", + " \"recall\": 0.32849365,\n", + " \"precision\": 0.6630037,\n", + " \"f1Score\": 0.4393204,\n", + " \"falsePositiveRate\": 0.022504892,\n", + " \"truePositiveCount\": \"181\",\n", + " \"falsePositiveCount\": \"92\",\n", + " \"falseNegativeCount\": \"370\",\n", + " \"trueNegativeCount\": \"3996\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.59818196,\n", + " \"recall\": 0.31941923,\n", + " \"precision\": 0.6743295,\n", + " \"f1Score\": 0.43349755,\n", + " \"falsePositiveRate\": 0.020792564,\n", + " \"truePositiveCount\": \"176\",\n", + " \"falsePositiveCount\": \"85\",\n", + " \"falseNegativeCount\": \"375\",\n", + " \"trueNegativeCount\": \"4003\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.60704446,\n", + " \"recall\": 0.30852994,\n", + " \"precision\": 0.6854839,\n", + " \"f1Score\": 0.42553192,\n", + " \"falsePositiveRate\": 0.019080235,\n", + " \"truePositiveCount\": \"170\",\n", + " \"falsePositiveCount\": \"78\",\n", + " \"falseNegativeCount\": \"381\",\n", + " \"trueNegativeCount\": \"4010\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6096991,\n", + " \"recall\": 0.29945552,\n", + " \"precision\": 0.6818182,\n", + " \"f1Score\": 0.41614124,\n", + " \"falsePositiveRate\": 0.018835617,\n", + " \"truePositiveCount\": \"165\",\n", + " \"falsePositiveCount\": \"77\",\n", + " \"falseNegativeCount\": \"386\",\n", + " \"trueNegativeCount\": \"4011\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6194289,\n", + " \"recall\": 0.28856623,\n", + " \"precision\": 0.6913043,\n", + " \"f1Score\": 0.4071703,\n", + " \"falsePositiveRate\": 0.017367907,\n", + " \"truePositiveCount\": \"159\",\n", + " \"falsePositiveCount\": \"71\",\n", + " \"falseNegativeCount\": \"392\",\n", + " \"trueNegativeCount\": \"4017\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6299003,\n", + " \"recall\": 0.27949184,\n", + " \"precision\": 0.6936937,\n", + " \"f1Score\": 0.3984476,\n", + " \"falsePositiveRate\": 0.01663405,\n", + " \"truePositiveCount\": \"154\",\n", + " \"falsePositiveCount\": \"68\",\n", + " \"falseNegativeCount\": \"397\",\n", + " \"trueNegativeCount\": \"4020\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6326805,\n", + " \"recall\": 0.26860255,\n", + " \"precision\": 0.6883721,\n", + " \"f1Score\": 0.38642296,\n", + " \"falsePositiveRate\": 0.016389433,\n", + " \"truePositiveCount\": \"148\",\n", + " \"falsePositiveCount\": \"67\",\n", + " \"falseNegativeCount\": \"403\",\n", + " \"trueNegativeCount\": \"4021\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6446329,\n", + " \"recall\": 0.25952813,\n", + " \"precision\": 0.71859294,\n", + " \"f1Score\": 0.38133332,\n", + " \"falsePositiveRate\": 0.01369863,\n", + " \"truePositiveCount\": \"143\",\n", + " \"falsePositiveCount\": \"56\",\n", + " \"falseNegativeCount\": \"408\",\n", + " \"trueNegativeCount\": \"4032\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6525863,\n", + " \"recall\": 0.24863884,\n", + " \"precision\": 0.7287234,\n", + " \"f1Score\": 0.37077132,\n", + " \"falsePositiveRate\": 0.012475538,\n", + " \"truePositiveCount\": \"137\",\n", + " \"falsePositiveCount\": \"51\",\n", + " \"falseNegativeCount\": \"414\",\n", + " \"trueNegativeCount\": \"4037\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6613302,\n", + " \"recall\": 0.23956443,\n", + " \"precision\": 0.73333335,\n", + " \"f1Score\": 0.3611491,\n", + " \"falsePositiveRate\": 0.011741683,\n", + " \"truePositiveCount\": \"132\",\n", + " \"falsePositiveCount\": \"48\",\n", + " \"falseNegativeCount\": \"419\",\n", + " \"trueNegativeCount\": \"4040\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6681088,\n", + " \"recall\": 0.22867514,\n", + " \"precision\": 0.73255813,\n", + " \"f1Score\": 0.34854773,\n", + " \"falsePositiveRate\": 0.011252446,\n", + " \"truePositiveCount\": \"126\",\n", + " \"falsePositiveCount\": \"46\",\n", + " \"falseNegativeCount\": \"425\",\n", + " \"trueNegativeCount\": \"4042\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6787454,\n", + " \"recall\": 0.21960072,\n", + " \"precision\": 0.7378049,\n", + " \"f1Score\": 0.33846155,\n", + " \"falsePositiveRate\": 0.010518591,\n", + " \"truePositiveCount\": \"121\",\n", + " \"falsePositiveCount\": \"43\",\n", + " \"falseNegativeCount\": \"430\",\n", + " \"trueNegativeCount\": \"4045\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.68742466,\n", + " \"recall\": 0.20871143,\n", + " \"precision\": 0.73717946,\n", + " \"f1Score\": 0.32531825,\n", + " \"falsePositiveRate\": 0.010029354,\n", + " \"truePositiveCount\": \"115\",\n", + " \"falsePositiveCount\": \"41\",\n", + " \"falseNegativeCount\": \"436\",\n", + " \"trueNegativeCount\": \"4047\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.69688004,\n", + " \"recall\": 0.19963703,\n", + " \"precision\": 0.7482993,\n", + " \"f1Score\": 0.31518623,\n", + " \"falsePositiveRate\": 0.009050881,\n", + " \"truePositiveCount\": \"110\",\n", + " \"falsePositiveCount\": \"37\",\n", + " \"falseNegativeCount\": \"441\",\n", + " \"trueNegativeCount\": \"4051\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.70484525,\n", + " \"recall\": 0.18874773,\n", + " \"precision\": 0.7647059,\n", + " \"f1Score\": 0.30276564,\n", + " \"falsePositiveRate\": 0.007827789,\n", + " \"truePositiveCount\": \"104\",\n", + " \"falsePositiveCount\": \"32\",\n", + " \"falseNegativeCount\": \"447\",\n", + " \"trueNegativeCount\": \"4056\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.71349317,\n", + " \"recall\": 0.17967331,\n", + " \"precision\": 0.76153845,\n", + " \"f1Score\": 0.2907489,\n", + " \"falsePositiveRate\": 0.00758317,\n", + " \"truePositiveCount\": \"99\",\n", + " \"falsePositiveCount\": \"31\",\n", + " \"falseNegativeCount\": \"452\",\n", + " \"trueNegativeCount\": \"4057\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.72145003,\n", + " \"recall\": 0.16878402,\n", + " \"precision\": 0.75609756,\n", + " \"f1Score\": 0.27596438,\n", + " \"falsePositiveRate\": 0.007338552,\n", + " \"truePositiveCount\": \"93\",\n", + " \"falsePositiveCount\": \"30\",\n", + " \"falseNegativeCount\": \"458\",\n", + " \"trueNegativeCount\": \"4058\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.72871804,\n", + " \"recall\": 0.15970962,\n", + " \"precision\": 0.76521736,\n", + " \"f1Score\": 0.26426426,\n", + " \"falsePositiveRate\": 0.0066046966,\n", + " \"truePositiveCount\": \"88\",\n", + " \"falsePositiveCount\": \"27\",\n", + " \"falseNegativeCount\": \"463\",\n", + " \"trueNegativeCount\": \"4061\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.74161375,\n", + " \"recall\": 0.14882033,\n", + " \"precision\": 0.76635516,\n", + " \"f1Score\": 0.24924012,\n", + " \"falsePositiveRate\": 0.00611546,\n", + " \"truePositiveCount\": \"82\",\n", + " \"falsePositiveCount\": \"25\",\n", + " \"falseNegativeCount\": \"469\",\n", + " \"trueNegativeCount\": \"4063\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.74766874,\n", + " \"recall\": 0.13974592,\n", + " \"precision\": 0.78571427,\n", + " \"f1Score\": 0.23728813,\n", + " \"falsePositiveRate\": 0.0051369863,\n", + " \"truePositiveCount\": \"77\",\n", + " \"falsePositiveCount\": \"21\",\n", + " \"falseNegativeCount\": \"474\",\n", + " \"trueNegativeCount\": \"4067\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.75845456,\n", + " \"recall\": 0.12885663,\n", + " \"precision\": 0.8068182,\n", + " \"f1Score\": 0.22222222,\n", + " \"falsePositiveRate\": 0.0041585127,\n", + " \"truePositiveCount\": \"71\",\n", + " \"falsePositiveCount\": \"17\",\n", + " \"falseNegativeCount\": \"480\",\n", + " \"trueNegativeCount\": \"4071\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.7666716,\n", + " \"recall\": 0.11978222,\n", + " \"precision\": 0.80487806,\n", + " \"f1Score\": 0.2085308,\n", + " \"falsePositiveRate\": 0.0039138943,\n", + " \"truePositiveCount\": \"66\",\n", + " \"falsePositiveCount\": \"16\",\n", + " \"falseNegativeCount\": \"485\",\n", + " \"trueNegativeCount\": \"4072\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.7772814,\n", + " \"recall\": 0.108892925,\n", + " \"precision\": 0.7894737,\n", + " \"f1Score\": 0.19138756,\n", + " \"falsePositiveRate\": 0.0039138943,\n", + " \"truePositiveCount\": \"60\",\n", + " \"falsePositiveCount\": \"16\",\n", + " \"falseNegativeCount\": \"491\",\n", + " \"trueNegativeCount\": \"4072\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.78661436,\n", + " \"recall\": 0.09981851,\n", + " \"precision\": 0.8088235,\n", + " \"f1Score\": 0.17770597,\n", + " \"falsePositiveRate\": 0.003180039,\n", + " \"truePositiveCount\": \"55\",\n", + " \"falsePositiveCount\": \"13\",\n", + " \"falseNegativeCount\": \"496\",\n", + " \"trueNegativeCount\": \"4075\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.8050728,\n", + " \"recall\": 0.08892922,\n", + " \"precision\": 0.8448276,\n", + " \"f1Score\": 0.16091955,\n", + " \"falsePositiveRate\": 0.0022015655,\n", + " \"truePositiveCount\": \"49\",\n", + " \"falsePositiveCount\": \"9\",\n", + " \"falseNegativeCount\": \"502\",\n", + " \"trueNegativeCount\": \"4079\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.8127946,\n", + " \"recall\": 0.07985481,\n", + " \"precision\": 0.8301887,\n", + " \"f1Score\": 0.14569536,\n", + " \"falsePositiveRate\": 0.0022015655,\n", + " \"truePositiveCount\": \"44\",\n", + " \"falsePositiveCount\": \"9\",\n", + " \"falseNegativeCount\": \"507\",\n", + " \"trueNegativeCount\": \"4079\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.82794523,\n", + " \"recall\": 0.06896552,\n", + " \"precision\": 0.82608694,\n", + " \"f1Score\": 0.12730318,\n", + " \"falsePositiveRate\": 0.0019569471,\n", + " \"truePositiveCount\": \"38\",\n", + " \"falsePositiveCount\": \"8\",\n", + " \"falseNegativeCount\": \"513\",\n", + " \"trueNegativeCount\": \"4080\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.83858025,\n", + " \"recall\": 0.05989111,\n", + " \"precision\": 0.84615386,\n", + " \"f1Score\": 0.11186441,\n", + " \"falsePositiveRate\": 0.0014677104,\n", + " \"truePositiveCount\": \"33\",\n", + " \"falsePositiveCount\": \"6\",\n", + " \"falseNegativeCount\": \"518\",\n", + " \"trueNegativeCount\": \"4082\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.85226774,\n", + " \"recall\": 0.049001817,\n", + " \"precision\": 0.87096775,\n", + " \"f1Score\": 0.0927835,\n", + " \"falsePositiveRate\": 0.0009784736,\n", + " \"truePositiveCount\": \"27\",\n", + " \"falsePositiveCount\": \"4\",\n", + " \"falseNegativeCount\": \"524\",\n", + " \"trueNegativeCount\": \"4084\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.86955357,\n", + " \"recall\": 0.039927404,\n", + " \"precision\": 0.88,\n", + " \"f1Score\": 0.07638889,\n", + " \"falsePositiveRate\": 0.0007338552,\n", + " \"truePositiveCount\": \"22\",\n", + " \"falsePositiveCount\": \"3\",\n", + " \"falseNegativeCount\": \"529\",\n", + " \"trueNegativeCount\": \"4085\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.8754055,\n", + " \"recall\": 0.029038113,\n", + " \"precision\": 0.84210527,\n", + " \"f1Score\": 0.056140352,\n", + " \"falsePositiveRate\": 0.0007338552,\n", + " \"truePositiveCount\": \"16\",\n", + " \"falsePositiveCount\": \"3\",\n", + " \"falseNegativeCount\": \"535\",\n", + " \"trueNegativeCount\": \"4085\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.888999,\n", + " \"recall\": 0.019963702,\n", + " \"precision\": 0.78571427,\n", + " \"f1Score\": 0.038938053,\n", + " \"falsePositiveRate\": 0.0007338552,\n", + " \"truePositiveCount\": \"11\",\n", + " \"falsePositiveCount\": \"3\",\n", + " \"falseNegativeCount\": \"540\",\n", + " \"trueNegativeCount\": \"4085\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.912919,\n", + " \"recall\": 0.00907441,\n", + " \"precision\": 0.71428573,\n", + " \"f1Score\": 0.017921148,\n", + " \"falsePositiveRate\": 0.0004892368,\n", + " \"truePositiveCount\": \"5\",\n", + " \"falsePositiveCount\": \"2\",\n", + " \"falseNegativeCount\": \"546\",\n", + " \"trueNegativeCount\": \"4086\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9486238,\n", + " \"precision\": \"NaN\",\n", + " \"f1Score\": \"NaN\",\n", + " \"falseNegativeCount\": \"551\",\n", + " \"trueNegativeCount\": \"4088\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 1.0,\n", + " \"precision\": \"NaN\",\n", + " \"f1Score\": \"NaN\",\n", + " \"falseNegativeCount\": \"551\",\n", + " \"trueNegativeCount\": \"4088\",\n", + " \"positionThreshold\": 2147483647\n", + " }\n", + " ],\n", + " \"auRoc\": 0.9330749,\n", + " \"logLoss\": 0.993795\n", + " },\n", + " \"displayName\": \"2\"\n", + " },\n", + " {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/TBL8310014324731019264/modelEvaluations/5840134723993193762\",\n", + " \"annotationSpecId\": \"not available\",\n", + " \"createTime\": \"2021-02-28T17:24:31.800758Z\",\n", + " \"evaluatedExampleCount\": 4088,\n", + " \"classificationEvaluationMetrics\": {\n", + " \"auPrc\": 0.9580479,\n", + " \"confidenceMetricsEntry\": [\n", + " {\n", + " \"recall\": 0.9640411,\n", + " \"precision\": 0.9307983,\n", + " \"f1Score\": 0.9471281,\n", + " \"falsePositiveRate\": 0.53176045,\n", + " \"truePositiveCount\": \"3941\",\n", + " \"falsePositiveCount\": \"293\",\n", + " \"falseNegativeCount\": \"147\",\n", + " \"trueNegativeCount\": \"258\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.07797722,\n", + " \"recall\": 0.9640411,\n", + " \"precision\": 0.9307983,\n", + " \"f1Score\": 0.9471281,\n", + " \"falsePositiveRate\": 0.53176045,\n", + " \"truePositiveCount\": \"3941\",\n", + " \"falsePositiveCount\": \"293\",\n", + " \"falseNegativeCount\": \"147\",\n", + " \"trueNegativeCount\": \"258\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.30998033,\n", + " \"recall\": 0.9640411,\n", + " \"precision\": 0.9307983,\n", + " \"f1Score\": 0.9471281,\n", + " \"falsePositiveRate\": 0.53176045,\n", + " \"truePositiveCount\": \"3941\",\n", + " \"falsePositiveCount\": \"293\",\n", + " \"falseNegativeCount\": \"147\",\n", + " \"trueNegativeCount\": \"258\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.39478266,\n", + " \"recall\": 0.9640411,\n", + " \"precision\": 0.9307983,\n", + " \"f1Score\": 0.9471281,\n", + " \"falsePositiveRate\": 0.53176045,\n", + " \"truePositiveCount\": \"3941\",\n", + " \"falsePositiveCount\": \"293\",\n", + " \"falseNegativeCount\": \"147\",\n", + " \"trueNegativeCount\": \"258\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.46149504,\n", + " \"recall\": 0.9640411,\n", + " \"precision\": 0.9307983,\n", + " \"f1Score\": 0.9471281,\n", + " \"falsePositiveRate\": 0.53176045,\n", + " \"truePositiveCount\": \"3941\",\n", + " \"falsePositiveCount\": \"293\",\n", + " \"falseNegativeCount\": \"147\",\n", + " \"trueNegativeCount\": \"258\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5,\n", + " \"recall\": 0.9640411,\n", + " \"precision\": 0.9307983,\n", + " \"f1Score\": 0.9471281,\n", + " \"falsePositiveRate\": 0.53176045,\n", + " \"truePositiveCount\": \"3941\",\n", + " \"falsePositiveCount\": \"293\",\n", + " \"falseNegativeCount\": \"147\",\n", + " \"trueNegativeCount\": \"258\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.51386535,\n", + " \"recall\": 0.95988256,\n", + " \"precision\": 0.93361884,\n", + " \"f1Score\": 0.94656855,\n", + " \"falsePositiveRate\": 0.50635207,\n", + " \"truePositiveCount\": \"3924\",\n", + " \"falsePositiveCount\": \"279\",\n", + " \"falseNegativeCount\": \"164\",\n", + " \"trueNegativeCount\": \"272\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5659395,\n", + " \"recall\": 0.94985324,\n", + " \"precision\": 0.9406492,\n", + " \"f1Score\": 0.9452288,\n", + " \"falsePositiveRate\": 0.4446461,\n", + " \"truePositiveCount\": \"3883\",\n", + " \"falsePositiveCount\": \"245\",\n", + " \"falseNegativeCount\": \"205\",\n", + " \"trueNegativeCount\": \"306\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6215324,\n", + " \"recall\": 0.93982387,\n", + " \"precision\": 0.94817376,\n", + " \"f1Score\": 0.94398034,\n", + " \"falsePositiveRate\": 0.38112524,\n", + " \"truePositiveCount\": \"3842\",\n", + " \"falsePositiveCount\": \"210\",\n", + " \"falseNegativeCount\": \"246\",\n", + " \"trueNegativeCount\": \"341\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.65884733,\n", + " \"recall\": 0.93003917,\n", + " \"precision\": 0.9533601,\n", + " \"f1Score\": 0.9415552,\n", + " \"falsePositiveRate\": 0.33756804,\n", + " \"truePositiveCount\": \"3802\",\n", + " \"falsePositiveCount\": \"186\",\n", + " \"falseNegativeCount\": \"286\",\n", + " \"trueNegativeCount\": \"365\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.7143411,\n", + " \"recall\": 0.9200098,\n", + " \"precision\": 0.9623849,\n", + " \"f1Score\": 0.9407204,\n", + " \"falsePositiveRate\": 0.26678765,\n", + " \"truePositiveCount\": \"3761\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"327\",\n", + " \"trueNegativeCount\": \"404\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.7492718,\n", + " \"recall\": 0.9099804,\n", + " \"precision\": 0.96749026,\n", + " \"f1Score\": 0.9378545,\n", + " \"falsePositiveRate\": 0.22686026,\n", + " \"truePositiveCount\": \"3720\",\n", + " \"falsePositiveCount\": \"125\",\n", + " \"falseNegativeCount\": \"368\",\n", + " \"trueNegativeCount\": \"426\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.7795446,\n", + " \"recall\": 0.9001957,\n", + " \"precision\": 0.97303015,\n", + " \"f1Score\": 0.93519694,\n", + " \"falsePositiveRate\": 0.18511796,\n", + " \"truePositiveCount\": \"3680\",\n", + " \"falsePositiveCount\": \"102\",\n", + " \"falseNegativeCount\": \"408\",\n", + " \"trueNegativeCount\": \"449\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.7989856,\n", + " \"recall\": 0.8899217,\n", + " \"precision\": 0.97559667,\n", + " \"f1Score\": 0.93079185,\n", + " \"falsePositiveRate\": 0.16515426,\n", + " \"truePositiveCount\": \"3638\",\n", + " \"falsePositiveCount\": \"91\",\n", + " \"falseNegativeCount\": \"450\",\n", + " \"trueNegativeCount\": \"460\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.82042044,\n", + " \"recall\": 0.87989235,\n", + " \"precision\": 0.9777113,\n", + " \"f1Score\": 0.9262263,\n", + " \"falsePositiveRate\": 0.14882033,\n", + " \"truePositiveCount\": \"3597\",\n", + " \"falsePositiveCount\": \"82\",\n", + " \"falseNegativeCount\": \"491\",\n", + " \"trueNegativeCount\": \"469\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.83562684,\n", + " \"recall\": 0.87010765,\n", + " \"precision\": 0.9790807,\n", + " \"f1Score\": 0.92138326,\n", + " \"falsePositiveRate\": 0.13793103,\n", + " \"truePositiveCount\": \"3557\",\n", + " \"falsePositiveCount\": \"76\",\n", + " \"falseNegativeCount\": \"531\",\n", + " \"trueNegativeCount\": \"475\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.85593396,\n", + " \"recall\": 0.8600783,\n", + " \"precision\": 0.98102677,\n", + " \"f1Score\": 0.9165798,\n", + " \"falsePositiveRate\": 0.123411976,\n", + " \"truePositiveCount\": \"3516\",\n", + " \"falsePositiveCount\": \"68\",\n", + " \"falseNegativeCount\": \"572\",\n", + " \"trueNegativeCount\": \"483\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.86959773,\n", + " \"recall\": 0.8500489,\n", + " \"precision\": 0.9810841,\n", + " \"f1Score\": 0.9108781,\n", + " \"falsePositiveRate\": 0.1215971,\n", + " \"truePositiveCount\": \"3475\",\n", + " \"falsePositiveCount\": \"67\",\n", + " \"falseNegativeCount\": \"613\",\n", + " \"trueNegativeCount\": \"484\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.88389677,\n", + " \"recall\": 0.83977497,\n", + " \"precision\": 0.98338586,\n", + " \"f1Score\": 0.90592426,\n", + " \"falsePositiveRate\": 0.10526316,\n", + " \"truePositiveCount\": \"3433\",\n", + " \"falsePositiveCount\": \"58\",\n", + " \"falseNegativeCount\": \"655\",\n", + " \"trueNegativeCount\": \"493\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.89513993,\n", + " \"recall\": 0.8299902,\n", + " \"precision\": 0.9840487,\n", + " \"f1Score\": 0.9004777,\n", + " \"falsePositiveRate\": 0.09981851,\n", + " \"truePositiveCount\": \"3393\",\n", + " \"falsePositiveCount\": \"55\",\n", + " \"falseNegativeCount\": \"695\",\n", + " \"trueNegativeCount\": \"496\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.90273213,\n", + " \"recall\": 0.82020545,\n", + " \"precision\": 0.9853071,\n", + " \"f1Score\": 0.8952076,\n", + " \"falsePositiveRate\": 0.0907441,\n", + " \"truePositiveCount\": \"3353\",\n", + " \"falsePositiveCount\": \"50\",\n", + " \"falseNegativeCount\": \"735\",\n", + " \"trueNegativeCount\": \"501\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9118437,\n", + " \"recall\": 0.81017613,\n", + " \"precision\": 0.98718333,\n", + " \"f1Score\": 0.88996375,\n", + " \"falsePositiveRate\": 0.07803993,\n", + " \"truePositiveCount\": \"3312\",\n", + " \"falsePositiveCount\": \"43\",\n", + " \"falseNegativeCount\": \"776\",\n", + " \"trueNegativeCount\": \"508\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9204901,\n", + " \"recall\": 0.79990214,\n", + " \"precision\": 0.98821396,\n", + " \"f1Score\": 0.8841422,\n", + " \"falsePositiveRate\": 0.0707804,\n", + " \"truePositiveCount\": \"3270\",\n", + " \"falsePositiveCount\": \"39\",\n", + " \"falseNegativeCount\": \"818\",\n", + " \"trueNegativeCount\": \"512\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9283992,\n", + " \"recall\": 0.7898728,\n", + " \"precision\": 0.9898835,\n", + " \"f1Score\": 0.87863946,\n", + " \"falsePositiveRate\": 0.05989111,\n", + " \"truePositiveCount\": \"3229\",\n", + " \"falsePositiveCount\": \"33\",\n", + " \"falseNegativeCount\": \"859\",\n", + " \"trueNegativeCount\": \"518\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.934707,\n", + " \"recall\": 0.77984345,\n", + " \"precision\": 0.9909854,\n", + " \"f1Score\": 0.8728268,\n", + " \"falsePositiveRate\": 0.05263158,\n", + " \"truePositiveCount\": \"3188\",\n", + " \"falsePositiveCount\": \"29\",\n", + " \"falseNegativeCount\": \"900\",\n", + " \"trueNegativeCount\": \"522\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9401358,\n", + " \"recall\": 0.7698141,\n", + " \"precision\": 0.99211854,\n", + " \"f1Score\": 0.86694217,\n", + " \"falsePositiveRate\": 0.04537205,\n", + " \"truePositiveCount\": \"3147\",\n", + " \"falsePositiveCount\": \"25\",\n", + " \"falseNegativeCount\": \"941\",\n", + " \"trueNegativeCount\": \"526\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.943966,\n", + " \"recall\": 0.7600294,\n", + " \"precision\": 0.9923347,\n", + " \"f1Score\": 0.86078405,\n", + " \"falsePositiveRate\": 0.043557167,\n", + " \"truePositiveCount\": \"3107\",\n", + " \"falsePositiveCount\": \"24\",\n", + " \"falseNegativeCount\": \"981\",\n", + " \"trueNegativeCount\": \"527\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9484665,\n", + " \"recall\": 0.7497554,\n", + " \"precision\": 0.9925518,\n", + " \"f1Score\": 0.85423636,\n", + " \"falsePositiveRate\": 0.041742288,\n", + " \"truePositiveCount\": \"3065\",\n", + " \"falsePositiveCount\": \"23\",\n", + " \"falseNegativeCount\": \"1023\",\n", + " \"trueNegativeCount\": \"528\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9512129,\n", + " \"recall\": 0.7399706,\n", + " \"precision\": 0.9931057,\n", + " \"f1Score\": 0.8480516,\n", + " \"falsePositiveRate\": 0.03811252,\n", + " \"truePositiveCount\": \"3025\",\n", + " \"falsePositiveCount\": \"21\",\n", + " \"falseNegativeCount\": \"1063\",\n", + " \"trueNegativeCount\": \"530\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.95533496,\n", + " \"recall\": 0.7299413,\n", + " \"precision\": 0.9946667,\n", + " \"f1Score\": 0.8419865,\n", + " \"falsePositiveRate\": 0.029038113,\n", + " \"truePositiveCount\": \"2984\",\n", + " \"falsePositiveCount\": \"16\",\n", + " \"falseNegativeCount\": \"1104\",\n", + " \"trueNegativeCount\": \"535\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.95724624,\n", + " \"recall\": 0.72015655,\n", + " \"precision\": 0.9956037,\n", + " \"f1Score\": 0.8357701,\n", + " \"falsePositiveRate\": 0.023593467,\n", + " \"truePositiveCount\": \"2944\",\n", + " \"falsePositiveCount\": \"13\",\n", + " \"falseNegativeCount\": \"1144\",\n", + " \"trueNegativeCount\": \"538\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.96008635,\n", + " \"recall\": 0.7101272,\n", + " \"precision\": 0.99588335,\n", + " \"f1Score\": 0.82907325,\n", + " \"falsePositiveRate\": 0.021778584,\n", + " \"truePositiveCount\": \"2903\",\n", + " \"falsePositiveCount\": \"12\",\n", + " \"falseNegativeCount\": \"1185\",\n", + " \"trueNegativeCount\": \"539\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9629221,\n", + " \"recall\": 0.70009786,\n", + " \"precision\": 0.99617124,\n", + " \"f1Score\": 0.82229567,\n", + " \"falsePositiveRate\": 0.019963702,\n", + " \"truePositiveCount\": \"2862\",\n", + " \"falsePositiveCount\": \"11\",\n", + " \"falseNegativeCount\": \"1226\",\n", + " \"trueNegativeCount\": \"540\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9646139,\n", + " \"recall\": 0.6900685,\n", + " \"precision\": 0.9961158,\n", + " \"f1Score\": 0.8153179,\n", + " \"falsePositiveRate\": 0.019963702,\n", + " \"truePositiveCount\": \"2821\",\n", + " \"falsePositiveCount\": \"11\",\n", + " \"falseNegativeCount\": \"1267\",\n", + " \"trueNegativeCount\": \"540\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.96656674,\n", + " \"recall\": 0.67979455,\n", + " \"precision\": 0.99605733,\n", + " \"f1Score\": 0.8080838,\n", + " \"falsePositiveRate\": 0.019963702,\n", + " \"truePositiveCount\": \"2779\",\n", + " \"falsePositiveCount\": \"11\",\n", + " \"falseNegativeCount\": \"1309\",\n", + " \"trueNegativeCount\": \"540\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.96897614,\n", + " \"recall\": 0.6700098,\n", + " \"precision\": 0.9963623,\n", + " \"f1Score\": 0.8012286,\n", + " \"falsePositiveRate\": 0.01814882,\n", + " \"truePositiveCount\": \"2739\",\n", + " \"falsePositiveCount\": \"10\",\n", + " \"falseNegativeCount\": \"1349\",\n", + " \"trueNegativeCount\": \"541\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9706901,\n", + " \"recall\": 0.6599804,\n", + " \"precision\": 0.99630725,\n", + " \"f1Score\": 0.79399645,\n", + " \"falsePositiveRate\": 0.01814882,\n", + " \"truePositiveCount\": \"2698\",\n", + " \"falsePositiveCount\": \"10\",\n", + " \"falseNegativeCount\": \"1390\",\n", + " \"trueNegativeCount\": \"541\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.97229564,\n", + " \"recall\": 0.6499511,\n", + " \"precision\": 0.99699813,\n", + " \"f1Score\": 0.7869095,\n", + " \"falsePositiveRate\": 0.014519056,\n", + " \"truePositiveCount\": \"2657\",\n", + " \"falsePositiveCount\": \"8\",\n", + " \"falseNegativeCount\": \"1431\",\n", + " \"trueNegativeCount\": \"543\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9738624,\n", + " \"recall\": 0.6399217,\n", + " \"precision\": 0.9969512,\n", + " \"f1Score\": 0.7794994,\n", + " \"falsePositiveRate\": 0.014519056,\n", + " \"truePositiveCount\": \"2616\",\n", + " \"falsePositiveCount\": \"8\",\n", + " \"falseNegativeCount\": \"1472\",\n", + " \"trueNegativeCount\": \"543\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.97582275,\n", + " \"recall\": 0.62989235,\n", + " \"precision\": 0.99728894,\n", + " \"f1Score\": 0.7721139,\n", + " \"falsePositiveRate\": 0.012704174,\n", + " \"truePositiveCount\": \"2575\",\n", + " \"falsePositiveCount\": \"7\",\n", + " \"falseNegativeCount\": \"1513\",\n", + " \"trueNegativeCount\": \"544\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9772092,\n", + " \"recall\": 0.62010765,\n", + " \"precision\": 0.9976387,\n", + " \"f1Score\": 0.76482123,\n", + " \"falsePositiveRate\": 0.010889292,\n", + " \"truePositiveCount\": \"2535\",\n", + " \"falsePositiveCount\": \"6\",\n", + " \"falseNegativeCount\": \"1553\",\n", + " \"trueNegativeCount\": \"545\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9781838,\n", + " \"recall\": 0.6100783,\n", + " \"precision\": 0.9979992,\n", + " \"f1Score\": 0.7572491,\n", + " \"falsePositiveRate\": 0.00907441,\n", + " \"truePositiveCount\": \"2494\",\n", + " \"falsePositiveCount\": \"5\",\n", + " \"falseNegativeCount\": \"1594\",\n", + " \"trueNegativeCount\": \"546\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9794321,\n", + " \"recall\": 0.5998043,\n", + " \"precision\": 0.99837136,\n", + " \"f1Score\": 0.74938875,\n", + " \"falsePositiveRate\": 0.007259528,\n", + " \"truePositiveCount\": \"2452\",\n", + " \"falsePositiveCount\": \"4\",\n", + " \"falseNegativeCount\": \"1636\",\n", + " \"trueNegativeCount\": \"547\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9804444,\n", + " \"recall\": 0.5900196,\n", + " \"precision\": 0.99834436,\n", + " \"f1Score\": 0.74169743,\n", + " \"falsePositiveRate\": 0.007259528,\n", + " \"truePositiveCount\": \"2412\",\n", + " \"falsePositiveCount\": \"4\",\n", + " \"falseNegativeCount\": \"1676\",\n", + " \"trueNegativeCount\": \"547\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9813577,\n", + " \"recall\": 0.5799902,\n", + " \"precision\": 0.9991572,\n", + " \"f1Score\": 0.7339421,\n", + " \"falsePositiveRate\": 0.003629764,\n", + " \"truePositiveCount\": \"2371\",\n", + " \"falsePositiveCount\": \"2\",\n", + " \"falseNegativeCount\": \"1717\",\n", + " \"trueNegativeCount\": \"549\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9822546,\n", + " \"recall\": 0.56996083,\n", + " \"precision\": 0.99914235,\n", + " \"f1Score\": 0.7258567,\n", + " \"falsePositiveRate\": 0.003629764,\n", + " \"truePositiveCount\": \"2330\",\n", + " \"falsePositiveCount\": \"2\",\n", + " \"falseNegativeCount\": \"1758\",\n", + " \"trueNegativeCount\": \"549\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9832565,\n", + " \"recall\": 0.5599315,\n", + " \"precision\": 0.99912703,\n", + " \"f1Score\": 0.71766734,\n", + " \"falsePositiveRate\": 0.003629764,\n", + " \"truePositiveCount\": \"2289\",\n", + " \"falsePositiveCount\": \"2\",\n", + " \"falseNegativeCount\": \"1799\",\n", + " \"trueNegativeCount\": \"549\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.983971,\n", + " \"recall\": 0.54990214,\n", + " \"precision\": 0.99955535,\n", + " \"f1Score\": 0.709484,\n", + " \"falsePositiveRate\": 0.001814882,\n", + " \"truePositiveCount\": \"2248\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"1840\",\n", + " \"trueNegativeCount\": \"550\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98478603,\n", + " \"recall\": 0.54011744,\n", + " \"precision\": 0.9995473,\n", + " \"f1Score\": 0.7012863,\n", + " \"falsePositiveRate\": 0.001814882,\n", + " \"truePositiveCount\": \"2208\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"1880\",\n", + " \"trueNegativeCount\": \"550\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98576885,\n", + " \"recall\": 0.52984345,\n", + " \"precision\": 0.99953854,\n", + " \"f1Score\": 0.6925659,\n", + " \"falsePositiveRate\": 0.001814882,\n", + " \"truePositiveCount\": \"2166\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"1922\",\n", + " \"trueNegativeCount\": \"550\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9865271,\n", + " \"recall\": 0.5198141,\n", + " \"precision\": 0.99952966,\n", + " \"f1Score\": 0.6839395,\n", + " \"falsePositiveRate\": 0.001814882,\n", + " \"truePositiveCount\": \"2125\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"1963\",\n", + " \"trueNegativeCount\": \"550\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98722655,\n", + " \"recall\": 0.5100294,\n", + " \"precision\": 0.9995206,\n", + " \"f1Score\": 0.675413,\n", + " \"falsePositiveRate\": 0.001814882,\n", + " \"truePositiveCount\": \"2085\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2003\",\n", + " \"trueNegativeCount\": \"550\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98776835,\n", + " \"recall\": 0.49975538,\n", + " \"precision\": 0.99951077,\n", + " \"f1Score\": 0.66634053,\n", + " \"falsePositiveRate\": 0.001814882,\n", + " \"truePositiveCount\": \"2043\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2045\",\n", + " \"trueNegativeCount\": \"550\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98843795,\n", + " \"recall\": 0.49021527,\n", + " \"precision\": 0.9995012,\n", + " \"f1Score\": 0.657804,\n", + " \"falsePositiveRate\": 0.001814882,\n", + " \"truePositiveCount\": \"2004\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2084\",\n", + " \"trueNegativeCount\": \"550\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98898214,\n", + " \"recall\": 0.4801859,\n", + " \"precision\": 0.99949086,\n", + " \"f1Score\": 0.64871114,\n", + " \"falsePositiveRate\": 0.001814882,\n", + " \"truePositiveCount\": \"1963\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2125\",\n", + " \"trueNegativeCount\": \"550\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98952276,\n", + " \"recall\": 0.46991193,\n", + " \"precision\": 0.9994797,\n", + " \"f1Score\": 0.63926786,\n", + " \"falsePositiveRate\": 0.001814882,\n", + " \"truePositiveCount\": \"1921\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2167\",\n", + " \"trueNegativeCount\": \"550\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99007314,\n", + " \"recall\": 0.4598826,\n", + " \"precision\": 0.9994684,\n", + " \"f1Score\": 0.62992126,\n", + " \"falsePositiveRate\": 0.001814882,\n", + " \"truePositiveCount\": \"1880\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2208\",\n", + " \"trueNegativeCount\": \"550\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.990457,\n", + " \"recall\": 0.45009786,\n", + " \"precision\": 0.9994568,\n", + " \"f1Score\": 0.620678,\n", + " \"falsePositiveRate\": 0.001814882,\n", + " \"truePositiveCount\": \"1840\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2248\",\n", + " \"trueNegativeCount\": \"550\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99079466,\n", + " \"recall\": 0.44006848,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.61117715,\n", + " \"truePositiveCount\": \"1799\",\n", + " \"falseNegativeCount\": \"2289\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9911339,\n", + " \"recall\": 0.43003914,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.60143685,\n", + " \"truePositiveCount\": \"1758\",\n", + " \"falseNegativeCount\": \"2330\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99141955,\n", + " \"recall\": 0.4200098,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.591559,\n", + " \"truePositiveCount\": \"1717\",\n", + " \"falseNegativeCount\": \"2371\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9917892,\n", + " \"recall\": 0.40998042,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.5815406,\n", + " \"truePositiveCount\": \"1676\",\n", + " \"falseNegativeCount\": \"2412\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9921363,\n", + " \"recall\": 0.39995107,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.57137865,\n", + " \"truePositiveCount\": \"1635\",\n", + " \"falseNegativeCount\": \"2453\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99253094,\n", + " \"recall\": 0.38992172,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.56107,\n", + " \"truePositiveCount\": \"1594\",\n", + " \"falseNegativeCount\": \"2494\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9928577,\n", + " \"recall\": 0.380137,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.5508685,\n", + " \"truePositiveCount\": \"1554\",\n", + " \"falseNegativeCount\": \"2534\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99313205,\n", + " \"recall\": 0.369863,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.54,\n", + " \"truePositiveCount\": \"1512\",\n", + " \"falseNegativeCount\": \"2576\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9933454,\n", + " \"recall\": 0.35983366,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.5292319,\n", + " \"truePositiveCount\": \"1471\",\n", + " \"falseNegativeCount\": \"2617\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99355274,\n", + " \"recall\": 0.35004893,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.5185722,\n", + " \"truePositiveCount\": \"1431\",\n", + " \"falseNegativeCount\": \"2657\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99378794,\n", + " \"recall\": 0.33977494,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.507212,\n", + " \"truePositiveCount\": \"1389\",\n", + " \"falseNegativeCount\": \"2699\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9939992,\n", + " \"recall\": 0.33023483,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.49650607,\n", + " \"truePositiveCount\": \"1350\",\n", + " \"falseNegativeCount\": \"2738\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9942213,\n", + " \"recall\": 0.31996086,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.48480356,\n", + " \"truePositiveCount\": \"1308\",\n", + " \"falseNegativeCount\": \"2780\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9944552,\n", + " \"recall\": 0.30993152,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.47320262,\n", + " \"truePositiveCount\": \"1267\",\n", + " \"falseNegativeCount\": \"2821\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9946339,\n", + " \"recall\": 0.29990214,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.46142265,\n", + " \"truePositiveCount\": \"1226\",\n", + " \"falseNegativeCount\": \"2862\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99486357,\n", + " \"recall\": 0.2898728,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.44945952,\n", + " \"truePositiveCount\": \"1185\",\n", + " \"falseNegativeCount\": \"2903\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9950511,\n", + " \"recall\": 0.28008807,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.4376075,\n", + " \"truePositiveCount\": \"1145\",\n", + " \"falseNegativeCount\": \"2943\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9953063,\n", + " \"recall\": 0.2698141,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.42496628,\n", + " \"truePositiveCount\": \"1103\",\n", + " \"falseNegativeCount\": \"2985\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99548787,\n", + " \"recall\": 0.26002935,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.4127354,\n", + " \"truePositiveCount\": \"1063\",\n", + " \"falseNegativeCount\": \"3025\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99564606,\n", + " \"recall\": 0.25024462,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.40031305,\n", + " \"truePositiveCount\": \"1023\",\n", + " \"falseNegativeCount\": \"3065\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9958494,\n", + " \"recall\": 0.23997064,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.3870586,\n", + " \"truePositiveCount\": \"981\",\n", + " \"falseNegativeCount\": \"3107\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9959897,\n", + " \"recall\": 0.2299413,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.37390614,\n", + " \"truePositiveCount\": \"940\",\n", + " \"falseNegativeCount\": \"3148\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9961461,\n", + " \"recall\": 0.21991193,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.3605374,\n", + " \"truePositiveCount\": \"899\",\n", + " \"falseNegativeCount\": \"3189\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99626255,\n", + " \"recall\": 0.20988259,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.346947,\n", + " \"truePositiveCount\": \"858\",\n", + " \"falseNegativeCount\": \"3230\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99635655,\n", + " \"recall\": 0.19985323,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.33312947,\n", + " \"truePositiveCount\": \"817\",\n", + " \"falseNegativeCount\": \"3271\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9964815,\n", + " \"recall\": 0.1900685,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.31942445,\n", + " \"truePositiveCount\": \"777\",\n", + " \"falseNegativeCount\": \"3311\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9966108,\n", + " \"recall\": 0.17979452,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.30478954,\n", + " \"truePositiveCount\": \"735\",\n", + " \"falseNegativeCount\": \"3353\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99673027,\n", + " \"recall\": 0.16976516,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.29025513,\n", + " \"truePositiveCount\": \"694\",\n", + " \"falseNegativeCount\": \"3394\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99681914,\n", + " \"recall\": 0.15998043,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.27583298,\n", + " \"truePositiveCount\": \"654\",\n", + " \"falseNegativeCount\": \"3434\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9969455,\n", + " \"recall\": 0.15019569,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.26116547,\n", + " \"truePositiveCount\": \"614\",\n", + " \"falseNegativeCount\": \"3474\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99705213,\n", + " \"recall\": 0.13992172,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.24549356,\n", + " \"truePositiveCount\": \"572\",\n", + " \"falseNegativeCount\": \"3516\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9971576,\n", + " \"recall\": 0.13013698,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.23030303,\n", + " \"truePositiveCount\": \"532\",\n", + " \"falseNegativeCount\": \"3556\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9972461,\n", + " \"recall\": 0.11986301,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.21406728,\n", + " \"truePositiveCount\": \"490\",\n", + " \"falseNegativeCount\": \"3598\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9973425,\n", + " \"recall\": 0.110078275,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.19832525,\n", + " \"truePositiveCount\": \"450\",\n", + " \"falseNegativeCount\": \"3638\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9974355,\n", + " \"recall\": 0.10004892,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.18189904,\n", + " \"truePositiveCount\": \"409\",\n", + " \"falseNegativeCount\": \"3679\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9975045,\n", + " \"recall\": 0.09001957,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.16517055,\n", + " \"truePositiveCount\": \"368\",\n", + " \"falseNegativeCount\": \"3720\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99760437,\n", + " \"recall\": 0.079990216,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.14813137,\n", + " \"truePositiveCount\": \"327\",\n", + " \"falseNegativeCount\": \"3761\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9976864,\n", + " \"recall\": 0.07020548,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.1312,\n", + " \"truePositiveCount\": \"287\",\n", + " \"falseNegativeCount\": \"3801\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9977548,\n", + " \"recall\": 0.059931505,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.11308562,\n", + " \"truePositiveCount\": \"245\",\n", + " \"falseNegativeCount\": \"3843\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99784184,\n", + " \"recall\": 0.05014677,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.095504306,\n", + " \"truePositiveCount\": \"205\",\n", + " \"falseNegativeCount\": \"3883\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9979226,\n", + " \"recall\": 0.040117417,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.07714017,\n", + " \"truePositiveCount\": \"164\",\n", + " \"falseNegativeCount\": \"3924\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9980229,\n", + " \"recall\": 0.030088063,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.058418427,\n", + " \"truePositiveCount\": \"123\",\n", + " \"falseNegativeCount\": \"3965\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9981275,\n", + " \"recall\": 0.01981409,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.03885824,\n", + " \"truePositiveCount\": \"81\",\n", + " \"falseNegativeCount\": \"4007\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9982374,\n", + " \"recall\": 0.009784736,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.019379845,\n", + " \"truePositiveCount\": \"40\",\n", + " \"falseNegativeCount\": \"4048\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9987167,\n", + " \"precision\": \"NaN\",\n", + " \"f1Score\": \"NaN\",\n", + " \"falseNegativeCount\": \"4088\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 1.0,\n", + " \"precision\": \"NaN\",\n", + " \"f1Score\": \"NaN\",\n", + " \"falseNegativeCount\": \"4088\",\n", + " \"trueNegativeCount\": \"551\",\n", + " \"positionThreshold\": 1\n", + " }\n", + " ],\n", + " \"auRoc\": 0.9325568,\n", + " \"logLoss\": 0.09388557\n", + " },\n", + " \"displayName\": \"1\"\n", + " },\n", + " {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/TBL8310014324731019264/modelEvaluations/6226491855750944430\",\n", + " \"annotationSpecId\": \"not available\",\n", + " \"createTime\": \"2021-02-28T17:24:31.800758Z\",\n", + " \"evaluatedExampleCount\": 551,\n", + " \"classificationEvaluationMetrics\": {\n", + " \"auPrc\": 0.34112322,\n", + " \"confidenceMetricsEntry\": [\n", + " {\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.009267787,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.022539802,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.03954769,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.045538254,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.050852664,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.063637815,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.07210142,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.07750216,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.09185968,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.09674809,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.10816234,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.12236822,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.1267207,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.15273023,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.1649416,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.18867286,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.19824302,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.20855853,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.2164171,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.2228882,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.2343938,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.24437484,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.24899487,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.25527748,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.2643137,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.27015588,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.2785622,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.29147997,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.30125153,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.30796623,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.32498023,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.33223236,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.3348615,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.33746937,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.34359238,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.3495217,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.35605764,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.36666742,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.37787727,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.39064476,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.39657614,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.39991632,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.41141164,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.4184669,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.42953506,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.43759817,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.44597256,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.44895133,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.46139872,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.47061405,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.48159525,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.48736593,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.4939781,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.4995014,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5,\n", + " \"recall\": 0.46823958,\n", + " \"precision\": 0.63703704,\n", + " \"f1Score\": 0.53974897,\n", + " \"falsePositiveRate\": 0.035958905,\n", + " \"truePositiveCount\": \"258\",\n", + " \"falsePositiveCount\": \"147\",\n", + " \"falseNegativeCount\": \"293\",\n", + " \"trueNegativeCount\": \"3941\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5025096,\n", + " \"recall\": 0.45916516,\n", + " \"precision\": 0.63408524,\n", + " \"f1Score\": 0.5326316,\n", + " \"falsePositiveRate\": 0.035714287,\n", + " \"truePositiveCount\": \"253\",\n", + " \"falsePositiveCount\": \"146\",\n", + " \"falseNegativeCount\": \"298\",\n", + " \"trueNegativeCount\": \"3942\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.51137525,\n", + " \"recall\": 0.44827586,\n", + " \"precision\": 0.63010204,\n", + " \"f1Score\": 0.52386004,\n", + " \"falsePositiveRate\": 0.035469666,\n", + " \"truePositiveCount\": \"247\",\n", + " \"falsePositiveCount\": \"145\",\n", + " \"falseNegativeCount\": \"304\",\n", + " \"trueNegativeCount\": \"3943\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.51734245,\n", + " \"recall\": 0.43920144,\n", + " \"precision\": 0.62857145,\n", + " \"f1Score\": 0.517094,\n", + " \"falsePositiveRate\": 0.03498043,\n", + " \"truePositiveCount\": \"242\",\n", + " \"falsePositiveCount\": \"143\",\n", + " \"falseNegativeCount\": \"309\",\n", + " \"trueNegativeCount\": \"3945\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5248498,\n", + " \"recall\": 0.42831215,\n", + " \"precision\": 0.631016,\n", + " \"f1Score\": 0.5102703,\n", + " \"falsePositiveRate\": 0.03375734,\n", + " \"truePositiveCount\": \"236\",\n", + " \"falsePositiveCount\": \"138\",\n", + " \"falseNegativeCount\": \"315\",\n", + " \"trueNegativeCount\": \"3950\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.53241414,\n", + " \"recall\": 0.41923776,\n", + " \"precision\": 0.64705884,\n", + " \"f1Score\": 0.5088106,\n", + " \"falsePositiveRate\": 0.030821918,\n", + " \"truePositiveCount\": \"231\",\n", + " \"falsePositiveCount\": \"126\",\n", + " \"falseNegativeCount\": \"320\",\n", + " \"trueNegativeCount\": \"3962\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5427009,\n", + " \"recall\": 0.40834847,\n", + " \"precision\": 0.64841497,\n", + " \"f1Score\": 0.5011136,\n", + " \"falsePositiveRate\": 0.029843444,\n", + " \"truePositiveCount\": \"225\",\n", + " \"falsePositiveCount\": \"122\",\n", + " \"falseNegativeCount\": \"326\",\n", + " \"trueNegativeCount\": \"3966\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5505563,\n", + " \"recall\": 0.39927405,\n", + " \"precision\": 0.6547619,\n", + " \"f1Score\": 0.4960541,\n", + " \"falsePositiveRate\": 0.028375734,\n", + " \"truePositiveCount\": \"220\",\n", + " \"falsePositiveCount\": \"116\",\n", + " \"falseNegativeCount\": \"331\",\n", + " \"trueNegativeCount\": \"3972\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.55334836,\n", + " \"recall\": 0.38838476,\n", + " \"precision\": 0.65045595,\n", + " \"f1Score\": 0.48636365,\n", + " \"falsePositiveRate\": 0.028131116,\n", + " \"truePositiveCount\": \"214\",\n", + " \"falsePositiveCount\": \"115\",\n", + " \"falseNegativeCount\": \"337\",\n", + " \"trueNegativeCount\": \"3973\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.55837405,\n", + " \"recall\": 0.37931034,\n", + " \"precision\": 0.6490683,\n", + " \"f1Score\": 0.4788087,\n", + " \"falsePositiveRate\": 0.02764188,\n", + " \"truePositiveCount\": \"209\",\n", + " \"falsePositiveCount\": \"113\",\n", + " \"falseNegativeCount\": \"342\",\n", + " \"trueNegativeCount\": \"3975\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5685674,\n", + " \"recall\": 0.36842105,\n", + " \"precision\": 0.650641,\n", + " \"f1Score\": 0.47045192,\n", + " \"falsePositiveRate\": 0.026663406,\n", + " \"truePositiveCount\": \"203\",\n", + " \"falsePositiveCount\": \"109\",\n", + " \"falseNegativeCount\": \"348\",\n", + " \"trueNegativeCount\": \"3979\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5732897,\n", + " \"recall\": 0.35934663,\n", + " \"precision\": 0.6578073,\n", + " \"f1Score\": 0.46478873,\n", + " \"falsePositiveRate\": 0.025195695,\n", + " \"truePositiveCount\": \"198\",\n", + " \"falsePositiveCount\": \"103\",\n", + " \"falseNegativeCount\": \"353\",\n", + " \"trueNegativeCount\": \"3985\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5798386,\n", + " \"recall\": 0.34845734,\n", + " \"precision\": 0.65753424,\n", + " \"f1Score\": 0.455516,\n", + " \"falsePositiveRate\": 0.02446184,\n", + " \"truePositiveCount\": \"192\",\n", + " \"falsePositiveCount\": \"100\",\n", + " \"falseNegativeCount\": \"359\",\n", + " \"trueNegativeCount\": \"3988\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5841479,\n", + " \"recall\": 0.33938295,\n", + " \"precision\": 0.6561403,\n", + " \"f1Score\": 0.4473684,\n", + " \"falsePositiveRate\": 0.023972603,\n", + " \"truePositiveCount\": \"187\",\n", + " \"falsePositiveCount\": \"98\",\n", + " \"falseNegativeCount\": \"364\",\n", + " \"trueNegativeCount\": \"3990\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.58893955,\n", + " \"recall\": 0.32849365,\n", + " \"precision\": 0.6630037,\n", + " \"f1Score\": 0.4393204,\n", + " \"falsePositiveRate\": 0.022504892,\n", + " \"truePositiveCount\": \"181\",\n", + " \"falsePositiveCount\": \"92\",\n", + " \"falseNegativeCount\": \"370\",\n", + " \"trueNegativeCount\": \"3996\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.59818196,\n", + " \"recall\": 0.31941923,\n", + " \"precision\": 0.6743295,\n", + " \"f1Score\": 0.43349755,\n", + " \"falsePositiveRate\": 0.020792564,\n", + " \"truePositiveCount\": \"176\",\n", + " \"falsePositiveCount\": \"85\",\n", + " \"falseNegativeCount\": \"375\",\n", + " \"trueNegativeCount\": \"4003\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.60704446,\n", + " \"recall\": 0.30852994,\n", + " \"precision\": 0.6854839,\n", + " \"f1Score\": 0.42553192,\n", + " \"falsePositiveRate\": 0.019080235,\n", + " \"truePositiveCount\": \"170\",\n", + " \"falsePositiveCount\": \"78\",\n", + " \"falseNegativeCount\": \"381\",\n", + " \"trueNegativeCount\": \"4010\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6096991,\n", + " \"recall\": 0.29945552,\n", + " \"precision\": 0.6818182,\n", + " \"f1Score\": 0.41614124,\n", + " \"falsePositiveRate\": 0.018835617,\n", + " \"truePositiveCount\": \"165\",\n", + " \"falsePositiveCount\": \"77\",\n", + " \"falseNegativeCount\": \"386\",\n", + " \"trueNegativeCount\": \"4011\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6194289,\n", + " \"recall\": 0.28856623,\n", + " \"precision\": 0.6913043,\n", + " \"f1Score\": 0.4071703,\n", + " \"falsePositiveRate\": 0.017367907,\n", + " \"truePositiveCount\": \"159\",\n", + " \"falsePositiveCount\": \"71\",\n", + " \"falseNegativeCount\": \"392\",\n", + " \"trueNegativeCount\": \"4017\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6299003,\n", + " \"recall\": 0.27949184,\n", + " \"precision\": 0.6936937,\n", + " \"f1Score\": 0.3984476,\n", + " \"falsePositiveRate\": 0.01663405,\n", + " \"truePositiveCount\": \"154\",\n", + " \"falsePositiveCount\": \"68\",\n", + " \"falseNegativeCount\": \"397\",\n", + " \"trueNegativeCount\": \"4020\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6326805,\n", + " \"recall\": 0.26860255,\n", + " \"precision\": 0.6883721,\n", + " \"f1Score\": 0.38642296,\n", + " \"falsePositiveRate\": 0.016389433,\n", + " \"truePositiveCount\": \"148\",\n", + " \"falsePositiveCount\": \"67\",\n", + " \"falseNegativeCount\": \"403\",\n", + " \"trueNegativeCount\": \"4021\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6446329,\n", + " \"recall\": 0.25952813,\n", + " \"precision\": 0.71859294,\n", + " \"f1Score\": 0.38133332,\n", + " \"falsePositiveRate\": 0.01369863,\n", + " \"truePositiveCount\": \"143\",\n", + " \"falsePositiveCount\": \"56\",\n", + " \"falseNegativeCount\": \"408\",\n", + " \"trueNegativeCount\": \"4032\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6525863,\n", + " \"recall\": 0.24863884,\n", + " \"precision\": 0.7287234,\n", + " \"f1Score\": 0.37077132,\n", + " \"falsePositiveRate\": 0.012475538,\n", + " \"truePositiveCount\": \"137\",\n", + " \"falsePositiveCount\": \"51\",\n", + " \"falseNegativeCount\": \"414\",\n", + " \"trueNegativeCount\": \"4037\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6613302,\n", + " \"recall\": 0.23956443,\n", + " \"precision\": 0.73333335,\n", + " \"f1Score\": 0.3611491,\n", + " \"falsePositiveRate\": 0.011741683,\n", + " \"truePositiveCount\": \"132\",\n", + " \"falsePositiveCount\": \"48\",\n", + " \"falseNegativeCount\": \"419\",\n", + " \"trueNegativeCount\": \"4040\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6681088,\n", + " \"recall\": 0.22867514,\n", + " \"precision\": 0.73255813,\n", + " \"f1Score\": 0.34854773,\n", + " \"falsePositiveRate\": 0.011252446,\n", + " \"truePositiveCount\": \"126\",\n", + " \"falsePositiveCount\": \"46\",\n", + " \"falseNegativeCount\": \"425\",\n", + " \"trueNegativeCount\": \"4042\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6787454,\n", + " \"recall\": 0.21960072,\n", + " \"precision\": 0.7378049,\n", + " \"f1Score\": 0.33846155,\n", + " \"falsePositiveRate\": 0.010518591,\n", + " \"truePositiveCount\": \"121\",\n", + " \"falsePositiveCount\": \"43\",\n", + " \"falseNegativeCount\": \"430\",\n", + " \"trueNegativeCount\": \"4045\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.68742466,\n", + " \"recall\": 0.20871143,\n", + " \"precision\": 0.73717946,\n", + " \"f1Score\": 0.32531825,\n", + " \"falsePositiveRate\": 0.010029354,\n", + " \"truePositiveCount\": \"115\",\n", + " \"falsePositiveCount\": \"41\",\n", + " \"falseNegativeCount\": \"436\",\n", + " \"trueNegativeCount\": \"4047\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.69688004,\n", + " \"recall\": 0.19963703,\n", + " \"precision\": 0.7482993,\n", + " \"f1Score\": 0.31518623,\n", + " \"falsePositiveRate\": 0.009050881,\n", + " \"truePositiveCount\": \"110\",\n", + " \"falsePositiveCount\": \"37\",\n", + " \"falseNegativeCount\": \"441\",\n", + " \"trueNegativeCount\": \"4051\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.70484525,\n", + " \"recall\": 0.18874773,\n", + " \"precision\": 0.7647059,\n", + " \"f1Score\": 0.30276564,\n", + " \"falsePositiveRate\": 0.007827789,\n", + " \"truePositiveCount\": \"104\",\n", + " \"falsePositiveCount\": \"32\",\n", + " \"falseNegativeCount\": \"447\",\n", + " \"trueNegativeCount\": \"4056\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.71349317,\n", + " \"recall\": 0.17967331,\n", + " \"precision\": 0.76153845,\n", + " \"f1Score\": 0.2907489,\n", + " \"falsePositiveRate\": 0.00758317,\n", + " \"truePositiveCount\": \"99\",\n", + " \"falsePositiveCount\": \"31\",\n", + " \"falseNegativeCount\": \"452\",\n", + " \"trueNegativeCount\": \"4057\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.72145003,\n", + " \"recall\": 0.16878402,\n", + " \"precision\": 0.75609756,\n", + " \"f1Score\": 0.27596438,\n", + " \"falsePositiveRate\": 0.007338552,\n", + " \"truePositiveCount\": \"93\",\n", + " \"falsePositiveCount\": \"30\",\n", + " \"falseNegativeCount\": \"458\",\n", + " \"trueNegativeCount\": \"4058\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.72871804,\n", + " \"recall\": 0.15970962,\n", + " \"precision\": 0.76521736,\n", + " \"f1Score\": 0.26426426,\n", + " \"falsePositiveRate\": 0.0066046966,\n", + " \"truePositiveCount\": \"88\",\n", + " \"falsePositiveCount\": \"27\",\n", + " \"falseNegativeCount\": \"463\",\n", + " \"trueNegativeCount\": \"4061\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.74161375,\n", + " \"recall\": 0.14882033,\n", + " \"precision\": 0.76635516,\n", + " \"f1Score\": 0.24924012,\n", + " \"falsePositiveRate\": 0.00611546,\n", + " \"truePositiveCount\": \"82\",\n", + " \"falsePositiveCount\": \"25\",\n", + " \"falseNegativeCount\": \"469\",\n", + " \"trueNegativeCount\": \"4063\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.74766874,\n", + " \"recall\": 0.13974592,\n", + " \"precision\": 0.78571427,\n", + " \"f1Score\": 0.23728813,\n", + " \"falsePositiveRate\": 0.0051369863,\n", + " \"truePositiveCount\": \"77\",\n", + " \"falsePositiveCount\": \"21\",\n", + " \"falseNegativeCount\": \"474\",\n", + " \"trueNegativeCount\": \"4067\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.75845456,\n", + " \"recall\": 0.12885663,\n", + " \"precision\": 0.8068182,\n", + " \"f1Score\": 0.22222222,\n", + " \"falsePositiveRate\": 0.0041585127,\n", + " \"truePositiveCount\": \"71\",\n", + " \"falsePositiveCount\": \"17\",\n", + " \"falseNegativeCount\": \"480\",\n", + " \"trueNegativeCount\": \"4071\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.7666716,\n", + " \"recall\": 0.11978222,\n", + " \"precision\": 0.80487806,\n", + " \"f1Score\": 0.2085308,\n", + " \"falsePositiveRate\": 0.0039138943,\n", + " \"truePositiveCount\": \"66\",\n", + " \"falsePositiveCount\": \"16\",\n", + " \"falseNegativeCount\": \"485\",\n", + " \"trueNegativeCount\": \"4072\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.7772814,\n", + " \"recall\": 0.108892925,\n", + " \"precision\": 0.7894737,\n", + " \"f1Score\": 0.19138756,\n", + " \"falsePositiveRate\": 0.0039138943,\n", + " \"truePositiveCount\": \"60\",\n", + " \"falsePositiveCount\": \"16\",\n", + " \"falseNegativeCount\": \"491\",\n", + " \"trueNegativeCount\": \"4072\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.78661436,\n", + " \"recall\": 0.09981851,\n", + " \"precision\": 0.8088235,\n", + " \"f1Score\": 0.17770597,\n", + " \"falsePositiveRate\": 0.003180039,\n", + " \"truePositiveCount\": \"55\",\n", + " \"falsePositiveCount\": \"13\",\n", + " \"falseNegativeCount\": \"496\",\n", + " \"trueNegativeCount\": \"4075\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.8050728,\n", + " \"recall\": 0.08892922,\n", + " \"precision\": 0.8448276,\n", + " \"f1Score\": 0.16091955,\n", + " \"falsePositiveRate\": 0.0022015655,\n", + " \"truePositiveCount\": \"49\",\n", + " \"falsePositiveCount\": \"9\",\n", + " \"falseNegativeCount\": \"502\",\n", + " \"trueNegativeCount\": \"4079\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.8127946,\n", + " \"recall\": 0.07985481,\n", + " \"precision\": 0.8301887,\n", + " \"f1Score\": 0.14569536,\n", + " \"falsePositiveRate\": 0.0022015655,\n", + " \"truePositiveCount\": \"44\",\n", + " \"falsePositiveCount\": \"9\",\n", + " \"falseNegativeCount\": \"507\",\n", + " \"trueNegativeCount\": \"4079\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.82794523,\n", + " \"recall\": 0.06896552,\n", + " \"precision\": 0.82608694,\n", + " \"f1Score\": 0.12730318,\n", + " \"falsePositiveRate\": 0.0019569471,\n", + " \"truePositiveCount\": \"38\",\n", + " \"falsePositiveCount\": \"8\",\n", + " \"falseNegativeCount\": \"513\",\n", + " \"trueNegativeCount\": \"4080\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.83858025,\n", + " \"recall\": 0.05989111,\n", + " \"precision\": 0.84615386,\n", + " \"f1Score\": 0.11186441,\n", + " \"falsePositiveRate\": 0.0014677104,\n", + " \"truePositiveCount\": \"33\",\n", + " \"falsePositiveCount\": \"6\",\n", + " \"falseNegativeCount\": \"518\",\n", + " \"trueNegativeCount\": \"4082\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.85226774,\n", + " \"recall\": 0.049001817,\n", + " \"precision\": 0.87096775,\n", + " \"f1Score\": 0.0927835,\n", + " \"falsePositiveRate\": 0.0009784736,\n", + " \"truePositiveCount\": \"27\",\n", + " \"falsePositiveCount\": \"4\",\n", + " \"falseNegativeCount\": \"524\",\n", + " \"trueNegativeCount\": \"4084\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.86955357,\n", + " \"recall\": 0.039927404,\n", + " \"precision\": 0.88,\n", + " \"f1Score\": 0.07638889,\n", + " \"falsePositiveRate\": 0.0007338552,\n", + " \"truePositiveCount\": \"22\",\n", + " \"falsePositiveCount\": \"3\",\n", + " \"falseNegativeCount\": \"529\",\n", + " \"trueNegativeCount\": \"4085\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.8754055,\n", + " \"recall\": 0.029038113,\n", + " \"precision\": 0.84210527,\n", + " \"f1Score\": 0.056140352,\n", + " \"falsePositiveRate\": 0.0007338552,\n", + " \"truePositiveCount\": \"16\",\n", + " \"falsePositiveCount\": \"3\",\n", + " \"falseNegativeCount\": \"535\",\n", + " \"trueNegativeCount\": \"4085\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.888999,\n", + " \"recall\": 0.019963702,\n", + " \"precision\": 0.78571427,\n", + " \"f1Score\": 0.038938053,\n", + " \"falsePositiveRate\": 0.0007338552,\n", + " \"truePositiveCount\": \"11\",\n", + " \"falsePositiveCount\": \"3\",\n", + " \"falseNegativeCount\": \"540\",\n", + " \"trueNegativeCount\": \"4085\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.912919,\n", + " \"recall\": 0.00907441,\n", + " \"precision\": 0.71428573,\n", + " \"f1Score\": 0.017921148,\n", + " \"falsePositiveRate\": 0.0004892368,\n", + " \"truePositiveCount\": \"5\",\n", + " \"falsePositiveCount\": \"2\",\n", + " \"falseNegativeCount\": \"546\",\n", + " \"trueNegativeCount\": \"4086\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9486238,\n", + " \"precision\": \"NaN\",\n", + " \"f1Score\": \"NaN\",\n", + " \"falseNegativeCount\": \"551\",\n", + " \"trueNegativeCount\": \"4088\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 1.0,\n", + " \"precision\": \"NaN\",\n", + " \"f1Score\": \"NaN\",\n", + " \"falseNegativeCount\": \"551\",\n", + " \"trueNegativeCount\": \"4088\",\n", + " \"positionThreshold\": 1\n", + " }\n", + " ],\n", + " \"auRoc\": 0.71778876,\n", + " \"logLoss\": 0.993795\n", + " },\n", + " \"displayName\": \"2\"\n", + " },\n", + " {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/TBL8310014324731019264/modelEvaluations/7171857336139058652\",\n", + " \"annotationSpecId\": \"not available\",\n", + " \"createTime\": \"2021-02-28T17:24:31.800758Z\",\n", + " \"evaluatedExampleCount\": 4639,\n", + " \"classificationEvaluationMetrics\": {\n", + " \"auPrc\": 0.97598535,\n", + " \"confidenceMetricsEntry\": [\n", + " {\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.5,\n", + " \"f1Score\": 0.6666667,\n", + " \"falsePositiveRate\": 1.0,\n", + " \"truePositiveCount\": \"4639\",\n", + " \"falsePositiveCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.009267787,\n", + " \"recall\": 0.9997844,\n", + " \"precision\": 0.6205512,\n", + " \"f1Score\": 0.7657888,\n", + " \"falsePositiveRate\": 0.6113387,\n", + " \"truePositiveCount\": \"4638\",\n", + " \"falsePositiveCount\": \"2836\",\n", + " \"falseNegativeCount\": \"1\",\n", + " \"trueNegativeCount\": \"1803\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.08771089,\n", + " \"recall\": 0.99029964,\n", + " \"precision\": 0.77640694,\n", + " \"f1Score\": 0.87040544,\n", + " \"falsePositiveRate\": 0.28519076,\n", + " \"truePositiveCount\": \"4594\",\n", + " \"falsePositiveCount\": \"1323\",\n", + " \"falseNegativeCount\": \"45\",\n", + " \"trueNegativeCount\": \"3316\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.18867286,\n", + " \"recall\": 0.98016816,\n", + " \"precision\": 0.82268864,\n", + " \"f1Score\": 0.89455044,\n", + " \"falsePositiveRate\": 0.21125242,\n", + " \"truePositiveCount\": \"4547\",\n", + " \"falsePositiveCount\": \"980\",\n", + " \"falseNegativeCount\": \"92\",\n", + " \"trueNegativeCount\": \"3659\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.24711765,\n", + " \"recall\": 0.9702522,\n", + " \"precision\": 0.841623,\n", + " \"f1Score\": 0.9013718,\n", + " \"falsePositiveRate\": 0.18258245,\n", + " \"truePositiveCount\": \"4501\",\n", + " \"falsePositiveCount\": \"847\",\n", + " \"falseNegativeCount\": \"138\",\n", + " \"trueNegativeCount\": \"3792\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.29667678,\n", + " \"recall\": 0.96012074,\n", + " \"precision\": 0.85358375,\n", + " \"f1Score\": 0.90372324,\n", + " \"falsePositiveRate\": 0.16469067,\n", + " \"truePositiveCount\": \"4454\",\n", + " \"falsePositiveCount\": \"764\",\n", + " \"falseNegativeCount\": \"185\",\n", + " \"trueNegativeCount\": \"3875\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.33910435,\n", + " \"recall\": 0.9499892,\n", + " \"precision\": 0.8619206,\n", + " \"f1Score\": 0.9038146,\n", + " \"falsePositiveRate\": 0.15218797,\n", + " \"truePositiveCount\": \"4407\",\n", + " \"falsePositiveCount\": \"706\",\n", + " \"falseNegativeCount\": \"232\",\n", + " \"trueNegativeCount\": \"3933\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.3722213,\n", + " \"recall\": 0.94028884,\n", + " \"precision\": 0.87031126,\n", + " \"f1Score\": 0.9039478,\n", + " \"falsePositiveRate\": 0.14011641,\n", + " \"truePositiveCount\": \"4362\",\n", + " \"falsePositiveCount\": \"650\",\n", + " \"falseNegativeCount\": \"277\",\n", + " \"trueNegativeCount\": \"3989\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.41141066,\n", + " \"recall\": 0.93015736,\n", + " \"precision\": 0.87971455,\n", + " \"f1Score\": 0.90423304,\n", + " \"falsePositiveRate\": 0.12718259,\n", + " \"truePositiveCount\": \"4315\",\n", + " \"falsePositiveCount\": \"590\",\n", + " \"falseNegativeCount\": \"324\",\n", + " \"trueNegativeCount\": \"4049\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.44597256,\n", + " \"recall\": 0.920457,\n", + " \"precision\": 0.88921285,\n", + " \"f1Score\": 0.9045652,\n", + " \"falsePositiveRate\": 0.11467989,\n", + " \"truePositiveCount\": \"4270\",\n", + " \"falsePositiveCount\": \"532\",\n", + " \"falseNegativeCount\": \"369\",\n", + " \"trueNegativeCount\": \"4107\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.48088387,\n", + " \"recall\": 0.91010994,\n", + " \"precision\": 0.89829785,\n", + " \"f1Score\": 0.9041653,\n", + " \"falsePositiveRate\": 0.10303945,\n", + " \"truePositiveCount\": \"4222\",\n", + " \"falsePositiveCount\": \"478\",\n", + " \"falseNegativeCount\": \"417\",\n", + " \"trueNegativeCount\": \"4161\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5,\n", + " \"recall\": 0.90515196,\n", + " \"precision\": 0.90515196,\n", + " \"f1Score\": 0.90515196,\n", + " \"falsePositiveRate\": 0.09484803,\n", + " \"truePositiveCount\": \"4199\",\n", + " \"falsePositiveCount\": \"440\",\n", + " \"falseNegativeCount\": \"440\",\n", + " \"trueNegativeCount\": \"4199\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.51082355,\n", + " \"recall\": 0.9006251,\n", + " \"precision\": 0.90688086,\n", + " \"f1Score\": 0.90374213,\n", + " \"falsePositiveRate\": 0.09247683,\n", + " \"truePositiveCount\": \"4178\",\n", + " \"falsePositiveCount\": \"429\",\n", + " \"falseNegativeCount\": \"461\",\n", + " \"trueNegativeCount\": \"4210\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5395309,\n", + " \"recall\": 0.8902781,\n", + " \"precision\": 0.9145261,\n", + " \"f1Score\": 0.9022392,\n", + " \"falsePositiveRate\": 0.083207585,\n", + " \"truePositiveCount\": \"4130\",\n", + " \"falsePositiveCount\": \"386\",\n", + " \"falseNegativeCount\": \"509\",\n", + " \"trueNegativeCount\": \"4253\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.569127,\n", + " \"recall\": 0.88014656,\n", + " \"precision\": 0.92083895,\n", + " \"f1Score\": 0.90003306,\n", + " \"falsePositiveRate\": 0.07566286,\n", + " \"truePositiveCount\": \"4083\",\n", + " \"falsePositiveCount\": \"351\",\n", + " \"falseNegativeCount\": \"556\",\n", + " \"trueNegativeCount\": \"4288\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.59819055,\n", + " \"recall\": 0.8700151,\n", + " \"precision\": 0.9284564,\n", + " \"f1Score\": 0.8982862,\n", + " \"falsePositiveRate\": 0.06704031,\n", + " \"truePositiveCount\": \"4036\",\n", + " \"falsePositiveCount\": \"311\",\n", + " \"falseNegativeCount\": \"603\",\n", + " \"trueNegativeCount\": \"4328\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6267493,\n", + " \"recall\": 0.86009914,\n", + " \"precision\": 0.9350832,\n", + " \"f1Score\": 0.8960252,\n", + " \"falsePositiveRate\": 0.059711143,\n", + " \"truePositiveCount\": \"3990\",\n", + " \"falsePositiveCount\": \"277\",\n", + " \"falseNegativeCount\": \"649\",\n", + " \"trueNegativeCount\": \"4362\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6483357,\n", + " \"recall\": 0.84996766,\n", + " \"precision\": 0.9401526,\n", + " \"f1Score\": 0.8927884,\n", + " \"falsePositiveRate\": 0.05410649,\n", + " \"truePositiveCount\": \"3943\",\n", + " \"falsePositiveCount\": \"251\",\n", + " \"falseNegativeCount\": \"696\",\n", + " \"trueNegativeCount\": \"4388\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.68742466,\n", + " \"recall\": 0.8404829,\n", + " \"precision\": 0.95051193,\n", + " \"f1Score\": 0.8921176,\n", + " \"falsePositiveRate\": 0.04375943,\n", + " \"truePositiveCount\": \"3899\",\n", + " \"falsePositiveCount\": \"203\",\n", + " \"falseNegativeCount\": \"740\",\n", + " \"trueNegativeCount\": \"4436\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.7178758,\n", + " \"recall\": 0.83035135,\n", + " \"precision\": 0.95630586,\n", + " \"f1Score\": 0.8888889,\n", + " \"falsePositiveRate\": 0.03793921,\n", + " \"truePositiveCount\": \"3852\",\n", + " \"falsePositiveCount\": \"176\",\n", + " \"falseNegativeCount\": \"787\",\n", + " \"trueNegativeCount\": \"4463\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.7454512,\n", + " \"recall\": 0.8200043,\n", + " \"precision\": 0.9620637,\n", + " \"f1Score\": 0.8853718,\n", + " \"falsePositiveRate\": 0.032334555,\n", + " \"truePositiveCount\": \"3804\",\n", + " \"falsePositiveCount\": \"150\",\n", + " \"falseNegativeCount\": \"835\",\n", + " \"trueNegativeCount\": \"4489\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.7702336,\n", + " \"recall\": 0.8105195,\n", + " \"precision\": 0.9685729,\n", + " \"f1Score\": 0.8825255,\n", + " \"falsePositiveRate\": 0.02629877,\n", + " \"truePositiveCount\": \"3760\",\n", + " \"falsePositiveCount\": \"122\",\n", + " \"falseNegativeCount\": \"879\",\n", + " \"trueNegativeCount\": \"4517\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.79181457,\n", + " \"recall\": 0.80038804,\n", + " \"precision\": 0.972244,\n", + " \"f1Score\": 0.87798536,\n", + " \"falsePositiveRate\": 0.022849752,\n", + " \"truePositiveCount\": \"3713\",\n", + " \"falsePositiveCount\": \"106\",\n", + " \"falseNegativeCount\": \"926\",\n", + " \"trueNegativeCount\": \"4533\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.80905926,\n", + " \"recall\": 0.7902565,\n", + " \"precision\": 0.9755189,\n", + " \"f1Score\": 0.873169,\n", + " \"falsePositiveRate\": 0.01983186,\n", + " \"truePositiveCount\": \"3666\",\n", + " \"falsePositiveCount\": \"92\",\n", + " \"falseNegativeCount\": \"973\",\n", + " \"trueNegativeCount\": \"4547\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.82475233,\n", + " \"recall\": 0.7803406,\n", + " \"precision\": 0.97574127,\n", + " \"f1Score\": 0.86716974,\n", + " \"falsePositiveRate\": 0.019400733,\n", + " \"truePositiveCount\": \"3620\",\n", + " \"falsePositiveCount\": \"90\",\n", + " \"falseNegativeCount\": \"1019\",\n", + " \"trueNegativeCount\": \"4549\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.84347624,\n", + " \"recall\": 0.7702091,\n", + " \"precision\": 0.97863597,\n", + " \"f1Score\": 0.86200243,\n", + " \"falsePositiveRate\": 0.01681397,\n", + " \"truePositiveCount\": \"3573\",\n", + " \"falsePositiveCount\": \"78\",\n", + " \"falseNegativeCount\": \"1066\",\n", + " \"trueNegativeCount\": \"4561\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.8612723,\n", + " \"recall\": 0.7602932,\n", + " \"precision\": 0.9799944,\n", + " \"f1Score\": 0.8562758,\n", + " \"falsePositiveRate\": 0.015520587,\n", + " \"truePositiveCount\": \"3527\",\n", + " \"falsePositiveCount\": \"72\",\n", + " \"falseNegativeCount\": \"1112\",\n", + " \"trueNegativeCount\": \"4567\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.87450475,\n", + " \"recall\": 0.75016165,\n", + " \"precision\": 0.9811108,\n", + " \"f1Score\": 0.8502321,\n", + " \"falsePositiveRate\": 0.014442768,\n", + " \"truePositiveCount\": \"3480\",\n", + " \"falsePositiveCount\": \"67\",\n", + " \"falseNegativeCount\": \"1159\",\n", + " \"trueNegativeCount\": \"4572\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.8876167,\n", + " \"recall\": 0.74003017,\n", + " \"precision\": 0.9825415,\n", + " \"f1Score\": 0.8442149,\n", + " \"falsePositiveRate\": 0.013149385,\n", + " \"truePositiveCount\": \"3433\",\n", + " \"falsePositiveCount\": \"61\",\n", + " \"falseNegativeCount\": \"1206\",\n", + " \"trueNegativeCount\": \"4578\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.89729506,\n", + " \"recall\": 0.7303298,\n", + " \"precision\": 0.98402554,\n", + " \"f1Score\": 0.8384063,\n", + " \"falsePositiveRate\": 0.011856004,\n", + " \"truePositiveCount\": \"3388\",\n", + " \"falsePositiveCount\": \"55\",\n", + " \"falseNegativeCount\": \"1251\",\n", + " \"trueNegativeCount\": \"4584\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9069939,\n", + " \"recall\": 0.72041386,\n", + " \"precision\": 0.9858407,\n", + " \"f1Score\": 0.8324823,\n", + " \"falsePositiveRate\": 0.010347057,\n", + " \"truePositiveCount\": \"3342\",\n", + " \"falsePositiveCount\": \"48\",\n", + " \"falseNegativeCount\": \"1297\",\n", + " \"trueNegativeCount\": \"4591\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.91544837,\n", + " \"recall\": 0.7102824,\n", + " \"precision\": 0.987118,\n", + " \"f1Score\": 0.8261251,\n", + " \"falsePositiveRate\": 0.009269239,\n", + " \"truePositiveCount\": \"3295\",\n", + " \"falsePositiveCount\": \"43\",\n", + " \"falseNegativeCount\": \"1344\",\n", + " \"trueNegativeCount\": \"4596\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9246296,\n", + " \"recall\": 0.700582,\n", + " \"precision\": 0.9896468,\n", + " \"f1Score\": 0.8203963,\n", + " \"falsePositiveRate\": 0.007329166,\n", + " \"truePositiveCount\": \"3250\",\n", + " \"falsePositiveCount\": \"34\",\n", + " \"falseNegativeCount\": \"1389\",\n", + " \"trueNegativeCount\": \"4605\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.932719,\n", + " \"recall\": 0.6900194,\n", + " \"precision\": 0.99071497,\n", + " \"f1Score\": 0.8134689,\n", + " \"falsePositiveRate\": 0.006466911,\n", + " \"truePositiveCount\": \"3201\",\n", + " \"falsePositiveCount\": \"30\",\n", + " \"falseNegativeCount\": \"1438\",\n", + " \"trueNegativeCount\": \"4609\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.93940556,\n", + " \"recall\": 0.6807502,\n", + " \"precision\": 0.9921458,\n", + " \"f1Score\": 0.80746615,\n", + " \"falsePositiveRate\": 0.0053890925,\n", + " \"truePositiveCount\": \"3158\",\n", + " \"falsePositiveCount\": \"25\",\n", + " \"falseNegativeCount\": \"1481\",\n", + " \"trueNegativeCount\": \"4614\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9436984,\n", + " \"recall\": 0.6704031,\n", + " \"precision\": 0.99234205,\n", + " \"f1Score\": 0.8002058,\n", + " \"falsePositiveRate\": 0.0051735286,\n", + " \"truePositiveCount\": \"3110\",\n", + " \"falsePositiveCount\": \"24\",\n", + " \"falseNegativeCount\": \"1529\",\n", + " \"trueNegativeCount\": \"4615\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9486017,\n", + " \"recall\": 0.6602716,\n", + " \"precision\": 0.992547,\n", + " \"f1Score\": 0.7930097,\n", + " \"falsePositiveRate\": 0.004957965,\n", + " \"truePositiveCount\": \"3063\",\n", + " \"falsePositiveCount\": \"23\",\n", + " \"falseNegativeCount\": \"1576\",\n", + " \"trueNegativeCount\": \"4616\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.95302504,\n", + " \"recall\": 0.6501401,\n", + " \"precision\": 0.9937397,\n", + " \"f1Score\": 0.78603077,\n", + " \"falsePositiveRate\": 0.0040957103,\n", + " \"truePositiveCount\": \"3016\",\n", + " \"falsePositiveCount\": \"19\",\n", + " \"falseNegativeCount\": \"1623\",\n", + " \"trueNegativeCount\": \"4620\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.95584136,\n", + " \"recall\": 0.6400086,\n", + " \"precision\": 0.99530673,\n", + " \"f1Score\": 0.7790606,\n", + " \"falsePositiveRate\": 0.003017892,\n", + " \"truePositiveCount\": \"2969\",\n", + " \"falsePositiveCount\": \"14\",\n", + " \"falseNegativeCount\": \"1670\",\n", + " \"trueNegativeCount\": \"4625\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.95881057,\n", + " \"recall\": 0.6300927,\n", + " \"precision\": 0.9955722,\n", + " \"f1Score\": 0.7717492,\n", + " \"falsePositiveRate\": 0.002802328,\n", + " \"truePositiveCount\": \"2923\",\n", + " \"falsePositiveCount\": \"13\",\n", + " \"falseNegativeCount\": \"1716\",\n", + " \"trueNegativeCount\": \"4626\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.96166563,\n", + " \"recall\": 0.62082344,\n", + " \"precision\": 0.9961951,\n", + " \"f1Score\": 0.76494026,\n", + " \"falsePositiveRate\": 0.0023712006,\n", + " \"truePositiveCount\": \"2880\",\n", + " \"falsePositiveCount\": \"11\",\n", + " \"falseNegativeCount\": \"1759\",\n", + " \"trueNegativeCount\": \"4628\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.96420497,\n", + " \"recall\": 0.61026084,\n", + " \"precision\": 0.9961295,\n", + " \"f1Score\": 0.75685066,\n", + " \"falsePositiveRate\": 0.0023712006,\n", + " \"truePositiveCount\": \"2831\",\n", + " \"falsePositiveCount\": \"11\",\n", + " \"falseNegativeCount\": \"1808\",\n", + " \"trueNegativeCount\": \"4628\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.96650803,\n", + " \"recall\": 0.5999138,\n", + " \"precision\": 0.996063,\n", + " \"f1Score\": 0.7488228,\n", + " \"falsePositiveRate\": 0.0023712006,\n", + " \"truePositiveCount\": \"2783\",\n", + " \"falsePositiveCount\": \"11\",\n", + " \"falseNegativeCount\": \"1856\",\n", + " \"trueNegativeCount\": \"4628\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.96897614,\n", + " \"recall\": 0.59042895,\n", + " \"precision\": 0.9963623,\n", + " \"f1Score\": 0.74147266,\n", + " \"falsePositiveRate\": 0.002155637,\n", + " \"truePositiveCount\": \"2739\",\n", + " \"falsePositiveCount\": \"10\",\n", + " \"falseNegativeCount\": \"1900\",\n", + " \"trueNegativeCount\": \"4629\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.97092766,\n", + " \"recall\": 0.5807286,\n", + " \"precision\": 0.99667037,\n", + " \"f1Score\": 0.73385996,\n", + " \"falsePositiveRate\": 0.0019400733,\n", + " \"truePositiveCount\": \"2694\",\n", + " \"falsePositiveCount\": \"9\",\n", + " \"falseNegativeCount\": \"1945\",\n", + " \"trueNegativeCount\": \"4630\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9727514,\n", + " \"recall\": 0.5699504,\n", + " \"precision\": 0.9969834,\n", + " \"f1Score\": 0.7252777,\n", + " \"falsePositiveRate\": 0.0017245096,\n", + " \"truePositiveCount\": \"2644\",\n", + " \"falsePositiveCount\": \"8\",\n", + " \"falseNegativeCount\": \"1995\",\n", + " \"trueNegativeCount\": \"4631\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9747563,\n", + " \"recall\": 0.56046563,\n", + " \"precision\": 0.99731493,\n", + " \"f1Score\": 0.7176373,\n", + " \"falsePositiveRate\": 0.001508946,\n", + " \"truePositiveCount\": \"2600\",\n", + " \"falsePositiveCount\": \"7\",\n", + " \"falseNegativeCount\": \"2039\",\n", + " \"trueNegativeCount\": \"4632\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9766838,\n", + " \"recall\": 0.549903,\n", + " \"precision\": 0.9976535,\n", + " \"f1Score\": 0.709005,\n", + " \"falsePositiveRate\": 0.0012933821,\n", + " \"truePositiveCount\": \"2551\",\n", + " \"falsePositiveCount\": \"6\",\n", + " \"falseNegativeCount\": \"2088\",\n", + " \"trueNegativeCount\": \"4633\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.97776866,\n", + " \"recall\": 0.54063374,\n", + " \"precision\": 0.99801034,\n", + " \"f1Score\": 0.7013423,\n", + " \"falsePositiveRate\": 0.0010778185,\n", + " \"truePositiveCount\": \"2508\",\n", + " \"falsePositiveCount\": \"5\",\n", + " \"falseNegativeCount\": \"2131\",\n", + " \"trueNegativeCount\": \"4634\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9792647,\n", + " \"recall\": 0.53050226,\n", + " \"precision\": 0.9979724,\n", + " \"f1Score\": 0.6927516,\n", + " \"falsePositiveRate\": 0.0010778185,\n", + " \"truePositiveCount\": \"2461\",\n", + " \"falsePositiveCount\": \"5\",\n", + " \"falseNegativeCount\": \"2178\",\n", + " \"trueNegativeCount\": \"4634\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9804138,\n", + " \"recall\": 0.5203708,\n", + " \"precision\": 0.99834573,\n", + " \"f1Score\": 0.6841434,\n", + " \"falsePositiveRate\": 0.0008622548,\n", + " \"truePositiveCount\": \"2414\",\n", + " \"falsePositiveCount\": \"4\",\n", + " \"falseNegativeCount\": \"2225\",\n", + " \"trueNegativeCount\": \"4635\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9814273,\n", + " \"recall\": 0.51045483,\n", + " \"precision\": 0.9991561,\n", + " \"f1Score\": 0.6757027,\n", + " \"falsePositiveRate\": 0.0004311274,\n", + " \"truePositiveCount\": \"2368\",\n", + " \"falsePositiveCount\": \"2\",\n", + " \"falseNegativeCount\": \"2271\",\n", + " \"trueNegativeCount\": \"4637\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98243207,\n", + " \"recall\": 0.5005389,\n", + " \"precision\": 0.9991394,\n", + " \"f1Score\": 0.6669539,\n", + " \"falsePositiveRate\": 0.0004311274,\n", + " \"truePositiveCount\": \"2322\",\n", + " \"falsePositiveCount\": \"2\",\n", + " \"falseNegativeCount\": \"2317\",\n", + " \"trueNegativeCount\": \"4637\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9835001,\n", + " \"recall\": 0.49019185,\n", + " \"precision\": 0.9995604,\n", + " \"f1Score\": 0.6577958,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"2274\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2365\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9844133,\n", + " \"recall\": 0.4804915,\n", + " \"precision\": 0.9995516,\n", + " \"f1Score\": 0.6490028,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"2229\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2410\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98528725,\n", + " \"recall\": 0.47079113,\n", + " \"precision\": 0.99954236,\n", + " \"f1Score\": 0.6400938,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"2184\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2455\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98630506,\n", + " \"recall\": 0.46001294,\n", + " \"precision\": 0.9995316,\n", + " \"f1Score\": 0.6300561,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"2134\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2505\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9871925,\n", + " \"recall\": 0.44988143,\n", + " \"precision\": 0.9995211,\n", + " \"f1Score\": 0.6204846,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"2087\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2552\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9877509,\n", + " \"recall\": 0.44082776,\n", + " \"precision\": 0.99951124,\n", + " \"f1Score\": 0.6118175,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"2045\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2594\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9885219,\n", + " \"recall\": 0.43048072,\n", + " \"precision\": 0.9994995,\n", + " \"f1Score\": 0.6017779,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"1997\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2642\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9892192,\n", + " \"recall\": 0.4199181,\n", + " \"precision\": 0.9994869,\n", + " \"f1Score\": 0.5913783,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"1948\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2691\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9897657,\n", + " \"recall\": 0.41000214,\n", + " \"precision\": 0.9994745,\n", + " \"f1Score\": 0.5814735,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"1902\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2737\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9902156,\n", + " \"recall\": 0.40051734,\n", + " \"precision\": 0.99946207,\n", + " \"f1Score\": 0.57186824,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"1858\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2781\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.990671,\n", + " \"recall\": 0.3901703,\n", + " \"precision\": 0.9994478,\n", + " \"f1Score\": 0.5612403,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"1810\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2829\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99106044,\n", + " \"recall\": 0.38025436,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.5509917,\n", + " \"truePositiveCount\": \"1764\",\n", + " \"falseNegativeCount\": \"2875\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99142385,\n", + " \"recall\": 0.36990732,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.5400472,\n", + " \"truePositiveCount\": \"1716\",\n", + " \"falseNegativeCount\": \"2923\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.991821,\n", + " \"recall\": 0.35999137,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.52940243,\n", + " \"truePositiveCount\": \"1670\",\n", + " \"falseNegativeCount\": \"2969\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99219143,\n", + " \"recall\": 0.35007545,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.5186013,\n", + " \"truePositiveCount\": \"1624\",\n", + " \"falseNegativeCount\": \"3015\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99265563,\n", + " \"recall\": 0.3401595,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.50764036,\n", + " \"truePositiveCount\": \"1578\",\n", + " \"falseNegativeCount\": \"3061\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9930375,\n", + " \"recall\": 0.33045915,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.49675956,\n", + " \"truePositiveCount\": \"1533\",\n", + " \"falseNegativeCount\": \"3106\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9932301,\n", + " \"recall\": 0.32054323,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.48547176,\n", + " \"truePositiveCount\": \"1487\",\n", + " \"falseNegativeCount\": \"3152\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99349546,\n", + " \"recall\": 0.31041172,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.47376212,\n", + " \"truePositiveCount\": \"1440\",\n", + " \"falseNegativeCount\": \"3199\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.993766,\n", + " \"recall\": 0.30006468,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.461615,\n", + " \"truePositiveCount\": \"1392\",\n", + " \"falseNegativeCount\": \"3247\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9940096,\n", + " \"recall\": 0.29014874,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.44979113,\n", + " \"truePositiveCount\": \"1346\",\n", + " \"falseNegativeCount\": \"3293\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9942971,\n", + " \"recall\": 0.27980167,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.4372579,\n", + " \"truePositiveCount\": \"1298\",\n", + " \"falseNegativeCount\": \"3341\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9945464,\n", + " \"recall\": 0.26988575,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.42505518,\n", + " \"truePositiveCount\": \"1252\",\n", + " \"falseNegativeCount\": \"3387\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.994734,\n", + " \"recall\": 0.2601854,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.41293192,\n", + " \"truePositiveCount\": \"1207\",\n", + " \"falseNegativeCount\": \"3432\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9949861,\n", + " \"recall\": 0.25026944,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.40034482,\n", + " \"truePositiveCount\": \"1161\",\n", + " \"falseNegativeCount\": \"3478\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99523896,\n", + " \"recall\": 0.2399224,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.38699582,\n", + " \"truePositiveCount\": \"1113\",\n", + " \"falseNegativeCount\": \"3526\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99545944,\n", + " \"recall\": 0.23043759,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.37456203,\n", + " \"truePositiveCount\": \"1069\",\n", + " \"falseNegativeCount\": \"3570\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9956529,\n", + " \"recall\": 0.22009054,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.36077738,\n", + " \"truePositiveCount\": \"1021\",\n", + " \"falseNegativeCount\": \"3618\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9958653,\n", + " \"recall\": 0.21039017,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.34764025,\n", + " \"truePositiveCount\": \"976\",\n", + " \"falseNegativeCount\": \"3663\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99604857,\n", + " \"recall\": 0.20025867,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.33369252,\n", + " \"truePositiveCount\": \"929\",\n", + " \"falseNegativeCount\": \"3710\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99617386,\n", + " \"recall\": 0.19012718,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.31950733,\n", + " \"truePositiveCount\": \"882\",\n", + " \"falseNegativeCount\": \"3757\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9963188,\n", + " \"recall\": 0.18042682,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.3056976,\n", + " \"truePositiveCount\": \"837\",\n", + " \"falseNegativeCount\": \"3802\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99644965,\n", + " \"recall\": 0.17029533,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.29102966,\n", + " \"truePositiveCount\": \"790\",\n", + " \"falseNegativeCount\": \"3849\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99657893,\n", + " \"recall\": 0.16059496,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.27674592,\n", + " \"truePositiveCount\": \"745\",\n", + " \"falseNegativeCount\": \"3894\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99672806,\n", + " \"recall\": 0.1502479,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.2612444,\n", + " \"truePositiveCount\": \"697\",\n", + " \"falseNegativeCount\": \"3942\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9968369,\n", + " \"recall\": 0.14033197,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.24612476,\n", + " \"truePositiveCount\": \"651\",\n", + " \"falseNegativeCount\": \"3988\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9969655,\n", + " \"recall\": 0.13041604,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.23073989,\n", + " \"truePositiveCount\": \"605\",\n", + " \"falseNegativeCount\": \"4034\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9970963,\n", + " \"recall\": 0.12028454,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.21473928,\n", + " \"truePositiveCount\": \"558\",\n", + " \"falseNegativeCount\": \"4081\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9972023,\n", + " \"recall\": 0.110799745,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.19949543,\n", + " \"truePositiveCount\": \"514\",\n", + " \"falseNegativeCount\": \"4125\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9973033,\n", + " \"recall\": 0.100668244,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.18292205,\n", + " \"truePositiveCount\": \"467\",\n", + " \"falseNegativeCount\": \"4172\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99741936,\n", + " \"recall\": 0.09053675,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.16604072,\n", + " \"truePositiveCount\": \"420\",\n", + " \"falseNegativeCount\": \"4219\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9974875,\n", + " \"recall\": 0.080620825,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.14921205,\n", + " \"truePositiveCount\": \"374\",\n", + " \"falseNegativeCount\": \"4265\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9976044,\n", + " \"recall\": 0.070273764,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.13131924,\n", + " \"truePositiveCount\": \"326\",\n", + " \"falseNegativeCount\": \"4313\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99769753,\n", + " \"recall\": 0.06014227,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.11346076,\n", + " \"truePositiveCount\": \"279\",\n", + " \"falseNegativeCount\": \"4360\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99778664,\n", + " \"recall\": 0.049795214,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.09486653,\n", + " \"truePositiveCount\": \"231\",\n", + " \"falseNegativeCount\": \"4408\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9978742,\n", + " \"recall\": 0.040525977,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.07789517,\n", + " \"truePositiveCount\": \"188\",\n", + " \"falseNegativeCount\": \"4451\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.997992,\n", + " \"recall\": 0.030394481,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.058995817,\n", + " \"truePositiveCount\": \"141\",\n", + " \"falseNegativeCount\": \"4498\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9980989,\n", + " \"recall\": 0.020262988,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.03972111,\n", + " \"truePositiveCount\": \"94\",\n", + " \"falseNegativeCount\": \"4545\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99821925,\n", + " \"recall\": 0.010347057,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.020482184,\n", + " \"truePositiveCount\": \"48\",\n", + " \"falseNegativeCount\": \"4591\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9987167,\n", + " \"precision\": \"NaN\",\n", + " \"f1Score\": \"NaN\",\n", + " \"falseNegativeCount\": \"4639\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 1.0,\n", + " \"precision\": \"NaN\",\n", + " \"f1Score\": \"NaN\",\n", + " \"falseNegativeCount\": \"4639\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 2147483647\n", + " }\n", + " ],\n", + " \"confusionMatrix\": {\n", + " \"row\": [\n", + " {\n", + " \"exampleCount\": [\n", + " 3941,\n", + " 147\n", + " ]\n", + " },\n", + " {\n", + " \"exampleCount\": [\n", + " 293,\n", + " 258\n", + " ]\n", + " }\n", + " ],\n", + " \"displayName\": [\n", + " \"1\",\n", + " \"2\"\n", + " ]\n", + " },\n", + " \"auRoc\": 0.9743269,\n", + " \"logLoss\": 0.20077285\n", + " }\n", + " }\n", + "]\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "i6T0bzuNJjkY" + }, + "source": [ + "### [projects.locations.models.modelEvaluations.get](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models.modelEvaluations/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Bvhd2ITWU7i7" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "vKz1dRGaU7i8" + }, + "outputs": [], + "source": [ + "request = clients[\"tables\"].get_model_evaluation(\n", + " model_evaluation_name=evaluation_slice\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "YOHFYa0rU7i8" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ycPBW1ZvU7i8", + "outputId": "e410231d-89fd-4e58-bde3-3038302fb64c", + "scrolled": true + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/TBL8310014324731019264/modelEvaluations/124023939996839080\",\n", + " \"annotationSpecId\": \"not available\",\n", + " \"createTime\": \"2021-02-28T17:24:31.800758Z\",\n", + " \"evaluatedExampleCount\": 4639,\n", + " \"classificationEvaluationMetrics\": {\n", + " \"auPrc\": 0.89604473,\n", + " \"confidenceMetricsEntry\": [\n", + " {\n", + " \"recall\": 0.90515196,\n", + " \"precision\": 0.90515196,\n", + " \"f1Score\": 0.90515196,\n", + " \"falsePositiveRate\": 0.09484803,\n", + " \"truePositiveCount\": \"4199\",\n", + " \"falsePositiveCount\": \"440\",\n", + " \"falseNegativeCount\": \"440\",\n", + " \"trueNegativeCount\": \"4199\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.009267787,\n", + " \"recall\": 0.90515196,\n", + " \"precision\": 0.90515196,\n", + " \"f1Score\": 0.90515196,\n", + " \"falsePositiveRate\": 0.09484803,\n", + " \"truePositiveCount\": \"4199\",\n", + " \"falsePositiveCount\": \"440\",\n", + " \"falseNegativeCount\": \"440\",\n", + " \"trueNegativeCount\": \"4199\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.08771089,\n", + " \"recall\": 0.90515196,\n", + " \"precision\": 0.90515196,\n", + " \"f1Score\": 0.90515196,\n", + " \"falsePositiveRate\": 0.09484803,\n", + " \"truePositiveCount\": \"4199\",\n", + " \"falsePositiveCount\": \"440\",\n", + " \"falseNegativeCount\": \"440\",\n", + " \"trueNegativeCount\": \"4199\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.18867286,\n", + " \"recall\": 0.90515196,\n", + " \"precision\": 0.90515196,\n", + " \"f1Score\": 0.90515196,\n", + " \"falsePositiveRate\": 0.09484803,\n", + " \"truePositiveCount\": \"4199\",\n", + " \"falsePositiveCount\": \"440\",\n", + " \"falseNegativeCount\": \"440\",\n", + " \"trueNegativeCount\": \"4199\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.24711765,\n", + " \"recall\": 0.90515196,\n", + " \"precision\": 0.90515196,\n", + " \"f1Score\": 0.90515196,\n", + " \"falsePositiveRate\": 0.09484803,\n", + " \"truePositiveCount\": \"4199\",\n", + " \"falsePositiveCount\": \"440\",\n", + " \"falseNegativeCount\": \"440\",\n", + " \"trueNegativeCount\": \"4199\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.29667678,\n", + " \"recall\": 0.90515196,\n", + " \"precision\": 0.90515196,\n", + " \"f1Score\": 0.90515196,\n", + " \"falsePositiveRate\": 0.09484803,\n", + " \"truePositiveCount\": \"4199\",\n", + " \"falsePositiveCount\": \"440\",\n", + " \"falseNegativeCount\": \"440\",\n", + " \"trueNegativeCount\": \"4199\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.33910435,\n", + " \"recall\": 0.90515196,\n", + " \"precision\": 0.90515196,\n", + " \"f1Score\": 0.90515196,\n", + " \"falsePositiveRate\": 0.09484803,\n", + " \"truePositiveCount\": \"4199\",\n", + " \"falsePositiveCount\": \"440\",\n", + " \"falseNegativeCount\": \"440\",\n", + " \"trueNegativeCount\": \"4199\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.3722213,\n", + " \"recall\": 0.90515196,\n", + " \"precision\": 0.90515196,\n", + " \"f1Score\": 0.90515196,\n", + " \"falsePositiveRate\": 0.09484803,\n", + " \"truePositiveCount\": \"4199\",\n", + " \"falsePositiveCount\": \"440\",\n", + " \"falseNegativeCount\": \"440\",\n", + " \"trueNegativeCount\": \"4199\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.41141066,\n", + " \"recall\": 0.90515196,\n", + " \"precision\": 0.90515196,\n", + " \"f1Score\": 0.90515196,\n", + " \"falsePositiveRate\": 0.09484803,\n", + " \"truePositiveCount\": \"4199\",\n", + " \"falsePositiveCount\": \"440\",\n", + " \"falseNegativeCount\": \"440\",\n", + " \"trueNegativeCount\": \"4199\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.44597256,\n", + " \"recall\": 0.90515196,\n", + " \"precision\": 0.90515196,\n", + " \"f1Score\": 0.90515196,\n", + " \"falsePositiveRate\": 0.09484803,\n", + " \"truePositiveCount\": \"4199\",\n", + " \"falsePositiveCount\": \"440\",\n", + " \"falseNegativeCount\": \"440\",\n", + " \"trueNegativeCount\": \"4199\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.48088387,\n", + " \"recall\": 0.90515196,\n", + " \"precision\": 0.90515196,\n", + " \"f1Score\": 0.90515196,\n", + " \"falsePositiveRate\": 0.09484803,\n", + " \"truePositiveCount\": \"4199\",\n", + " \"falsePositiveCount\": \"440\",\n", + " \"falseNegativeCount\": \"440\",\n", + " \"trueNegativeCount\": \"4199\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5,\n", + " \"recall\": 0.90515196,\n", + " \"precision\": 0.90515196,\n", + " \"f1Score\": 0.90515196,\n", + " \"falsePositiveRate\": 0.09484803,\n", + " \"truePositiveCount\": \"4199\",\n", + " \"falsePositiveCount\": \"440\",\n", + " \"falseNegativeCount\": \"440\",\n", + " \"trueNegativeCount\": \"4199\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.51082355,\n", + " \"recall\": 0.9006251,\n", + " \"precision\": 0.90688086,\n", + " \"f1Score\": 0.90374213,\n", + " \"falsePositiveRate\": 0.09247683,\n", + " \"truePositiveCount\": \"4178\",\n", + " \"falsePositiveCount\": \"429\",\n", + " \"falseNegativeCount\": \"461\",\n", + " \"trueNegativeCount\": \"4210\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.5395309,\n", + " \"recall\": 0.8902781,\n", + " \"precision\": 0.9145261,\n", + " \"f1Score\": 0.9022392,\n", + " \"falsePositiveRate\": 0.083207585,\n", + " \"truePositiveCount\": \"4130\",\n", + " \"falsePositiveCount\": \"386\",\n", + " \"falseNegativeCount\": \"509\",\n", + " \"trueNegativeCount\": \"4253\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.569127,\n", + " \"recall\": 0.88014656,\n", + " \"precision\": 0.92083895,\n", + " \"f1Score\": 0.90003306,\n", + " \"falsePositiveRate\": 0.07566286,\n", + " \"truePositiveCount\": \"4083\",\n", + " \"falsePositiveCount\": \"351\",\n", + " \"falseNegativeCount\": \"556\",\n", + " \"trueNegativeCount\": \"4288\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.59819055,\n", + " \"recall\": 0.8700151,\n", + " \"precision\": 0.9284564,\n", + " \"f1Score\": 0.8982862,\n", + " \"falsePositiveRate\": 0.06704031,\n", + " \"truePositiveCount\": \"4036\",\n", + " \"falsePositiveCount\": \"311\",\n", + " \"falseNegativeCount\": \"603\",\n", + " \"trueNegativeCount\": \"4328\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6267493,\n", + " \"recall\": 0.86009914,\n", + " \"precision\": 0.9350832,\n", + " \"f1Score\": 0.8960252,\n", + " \"falsePositiveRate\": 0.059711143,\n", + " \"truePositiveCount\": \"3990\",\n", + " \"falsePositiveCount\": \"277\",\n", + " \"falseNegativeCount\": \"649\",\n", + " \"trueNegativeCount\": \"4362\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.6483357,\n", + " \"recall\": 0.84996766,\n", + " \"precision\": 0.9401526,\n", + " \"f1Score\": 0.8927884,\n", + " \"falsePositiveRate\": 0.05410649,\n", + " \"truePositiveCount\": \"3943\",\n", + " \"falsePositiveCount\": \"251\",\n", + " \"falseNegativeCount\": \"696\",\n", + " \"trueNegativeCount\": \"4388\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.68742466,\n", + " \"recall\": 0.8404829,\n", + " \"precision\": 0.95051193,\n", + " \"f1Score\": 0.8921176,\n", + " \"falsePositiveRate\": 0.04375943,\n", + " \"truePositiveCount\": \"3899\",\n", + " \"falsePositiveCount\": \"203\",\n", + " \"falseNegativeCount\": \"740\",\n", + " \"trueNegativeCount\": \"4436\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.7178758,\n", + " \"recall\": 0.83035135,\n", + " \"precision\": 0.95630586,\n", + " \"f1Score\": 0.8888889,\n", + " \"falsePositiveRate\": 0.03793921,\n", + " \"truePositiveCount\": \"3852\",\n", + " \"falsePositiveCount\": \"176\",\n", + " \"falseNegativeCount\": \"787\",\n", + " \"trueNegativeCount\": \"4463\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.7454512,\n", + " \"recall\": 0.8200043,\n", + " \"precision\": 0.9620637,\n", + " \"f1Score\": 0.8853718,\n", + " \"falsePositiveRate\": 0.032334555,\n", + " \"truePositiveCount\": \"3804\",\n", + " \"falsePositiveCount\": \"150\",\n", + " \"falseNegativeCount\": \"835\",\n", + " \"trueNegativeCount\": \"4489\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.7702336,\n", + " \"recall\": 0.8105195,\n", + " \"precision\": 0.9685729,\n", + " \"f1Score\": 0.8825255,\n", + " \"falsePositiveRate\": 0.02629877,\n", + " \"truePositiveCount\": \"3760\",\n", + " \"falsePositiveCount\": \"122\",\n", + " \"falseNegativeCount\": \"879\",\n", + " \"trueNegativeCount\": \"4517\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.79181457,\n", + " \"recall\": 0.80038804,\n", + " \"precision\": 0.972244,\n", + " \"f1Score\": 0.87798536,\n", + " \"falsePositiveRate\": 0.022849752,\n", + " \"truePositiveCount\": \"3713\",\n", + " \"falsePositiveCount\": \"106\",\n", + " \"falseNegativeCount\": \"926\",\n", + " \"trueNegativeCount\": \"4533\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.80905926,\n", + " \"recall\": 0.7902565,\n", + " \"precision\": 0.9755189,\n", + " \"f1Score\": 0.873169,\n", + " \"falsePositiveRate\": 0.01983186,\n", + " \"truePositiveCount\": \"3666\",\n", + " \"falsePositiveCount\": \"92\",\n", + " \"falseNegativeCount\": \"973\",\n", + " \"trueNegativeCount\": \"4547\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.82475233,\n", + " \"recall\": 0.7803406,\n", + " \"precision\": 0.97574127,\n", + " \"f1Score\": 0.86716974,\n", + " \"falsePositiveRate\": 0.019400733,\n", + " \"truePositiveCount\": \"3620\",\n", + " \"falsePositiveCount\": \"90\",\n", + " \"falseNegativeCount\": \"1019\",\n", + " \"trueNegativeCount\": \"4549\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.84347624,\n", + " \"recall\": 0.7702091,\n", + " \"precision\": 0.97863597,\n", + " \"f1Score\": 0.86200243,\n", + " \"falsePositiveRate\": 0.01681397,\n", + " \"truePositiveCount\": \"3573\",\n", + " \"falsePositiveCount\": \"78\",\n", + " \"falseNegativeCount\": \"1066\",\n", + " \"trueNegativeCount\": \"4561\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.8612723,\n", + " \"recall\": 0.7602932,\n", + " \"precision\": 0.9799944,\n", + " \"f1Score\": 0.8562758,\n", + " \"falsePositiveRate\": 0.015520587,\n", + " \"truePositiveCount\": \"3527\",\n", + " \"falsePositiveCount\": \"72\",\n", + " \"falseNegativeCount\": \"1112\",\n", + " \"trueNegativeCount\": \"4567\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.87450475,\n", + " \"recall\": 0.75016165,\n", + " \"precision\": 0.9811108,\n", + " \"f1Score\": 0.8502321,\n", + " \"falsePositiveRate\": 0.014442768,\n", + " \"truePositiveCount\": \"3480\",\n", + " \"falsePositiveCount\": \"67\",\n", + " \"falseNegativeCount\": \"1159\",\n", + " \"trueNegativeCount\": \"4572\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.8876167,\n", + " \"recall\": 0.74003017,\n", + " \"precision\": 0.9825415,\n", + " \"f1Score\": 0.8442149,\n", + " \"falsePositiveRate\": 0.013149385,\n", + " \"truePositiveCount\": \"3433\",\n", + " \"falsePositiveCount\": \"61\",\n", + " \"falseNegativeCount\": \"1206\",\n", + " \"trueNegativeCount\": \"4578\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.89729506,\n", + " \"recall\": 0.7303298,\n", + " \"precision\": 0.98402554,\n", + " \"f1Score\": 0.8384063,\n", + " \"falsePositiveRate\": 0.011856004,\n", + " \"truePositiveCount\": \"3388\",\n", + " \"falsePositiveCount\": \"55\",\n", + " \"falseNegativeCount\": \"1251\",\n", + " \"trueNegativeCount\": \"4584\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9069939,\n", + " \"recall\": 0.72041386,\n", + " \"precision\": 0.9858407,\n", + " \"f1Score\": 0.8324823,\n", + " \"falsePositiveRate\": 0.010347057,\n", + " \"truePositiveCount\": \"3342\",\n", + " \"falsePositiveCount\": \"48\",\n", + " \"falseNegativeCount\": \"1297\",\n", + " \"trueNegativeCount\": \"4591\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.91544837,\n", + " \"recall\": 0.7102824,\n", + " \"precision\": 0.987118,\n", + " \"f1Score\": 0.8261251,\n", + " \"falsePositiveRate\": 0.009269239,\n", + " \"truePositiveCount\": \"3295\",\n", + " \"falsePositiveCount\": \"43\",\n", + " \"falseNegativeCount\": \"1344\",\n", + " \"trueNegativeCount\": \"4596\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9246296,\n", + " \"recall\": 0.700582,\n", + " \"precision\": 0.9896468,\n", + " \"f1Score\": 0.8203963,\n", + " \"falsePositiveRate\": 0.007329166,\n", + " \"truePositiveCount\": \"3250\",\n", + " \"falsePositiveCount\": \"34\",\n", + " \"falseNegativeCount\": \"1389\",\n", + " \"trueNegativeCount\": \"4605\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.932719,\n", + " \"recall\": 0.6900194,\n", + " \"precision\": 0.99071497,\n", + " \"f1Score\": 0.8134689,\n", + " \"falsePositiveRate\": 0.006466911,\n", + " \"truePositiveCount\": \"3201\",\n", + " \"falsePositiveCount\": \"30\",\n", + " \"falseNegativeCount\": \"1438\",\n", + " \"trueNegativeCount\": \"4609\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.93940556,\n", + " \"recall\": 0.6807502,\n", + " \"precision\": 0.9921458,\n", + " \"f1Score\": 0.80746615,\n", + " \"falsePositiveRate\": 0.0053890925,\n", + " \"truePositiveCount\": \"3158\",\n", + " \"falsePositiveCount\": \"25\",\n", + " \"falseNegativeCount\": \"1481\",\n", + " \"trueNegativeCount\": \"4614\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9436984,\n", + " \"recall\": 0.6704031,\n", + " \"precision\": 0.99234205,\n", + " \"f1Score\": 0.8002058,\n", + " \"falsePositiveRate\": 0.0051735286,\n", + " \"truePositiveCount\": \"3110\",\n", + " \"falsePositiveCount\": \"24\",\n", + " \"falseNegativeCount\": \"1529\",\n", + " \"trueNegativeCount\": \"4615\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9486017,\n", + " \"recall\": 0.6602716,\n", + " \"precision\": 0.992547,\n", + " \"f1Score\": 0.7930097,\n", + " \"falsePositiveRate\": 0.004957965,\n", + " \"truePositiveCount\": \"3063\",\n", + " \"falsePositiveCount\": \"23\",\n", + " \"falseNegativeCount\": \"1576\",\n", + " \"trueNegativeCount\": \"4616\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.95302504,\n", + " \"recall\": 0.6501401,\n", + " \"precision\": 0.9937397,\n", + " \"f1Score\": 0.78603077,\n", + " \"falsePositiveRate\": 0.0040957103,\n", + " \"truePositiveCount\": \"3016\",\n", + " \"falsePositiveCount\": \"19\",\n", + " \"falseNegativeCount\": \"1623\",\n", + " \"trueNegativeCount\": \"4620\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.95584136,\n", + " \"recall\": 0.6400086,\n", + " \"precision\": 0.99530673,\n", + " \"f1Score\": 0.7790606,\n", + " \"falsePositiveRate\": 0.003017892,\n", + " \"truePositiveCount\": \"2969\",\n", + " \"falsePositiveCount\": \"14\",\n", + " \"falseNegativeCount\": \"1670\",\n", + " \"trueNegativeCount\": \"4625\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.95881057,\n", + " \"recall\": 0.6300927,\n", + " \"precision\": 0.9955722,\n", + " \"f1Score\": 0.7717492,\n", + " \"falsePositiveRate\": 0.002802328,\n", + " \"truePositiveCount\": \"2923\",\n", + " \"falsePositiveCount\": \"13\",\n", + " \"falseNegativeCount\": \"1716\",\n", + " \"trueNegativeCount\": \"4626\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.96166563,\n", + " \"recall\": 0.62082344,\n", + " \"precision\": 0.9961951,\n", + " \"f1Score\": 0.76494026,\n", + " \"falsePositiveRate\": 0.0023712006,\n", + " \"truePositiveCount\": \"2880\",\n", + " \"falsePositiveCount\": \"11\",\n", + " \"falseNegativeCount\": \"1759\",\n", + " \"trueNegativeCount\": \"4628\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.96420497,\n", + " \"recall\": 0.61026084,\n", + " \"precision\": 0.9961295,\n", + " \"f1Score\": 0.75685066,\n", + " \"falsePositiveRate\": 0.0023712006,\n", + " \"truePositiveCount\": \"2831\",\n", + " \"falsePositiveCount\": \"11\",\n", + " \"falseNegativeCount\": \"1808\",\n", + " \"trueNegativeCount\": \"4628\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.96650803,\n", + " \"recall\": 0.5999138,\n", + " \"precision\": 0.996063,\n", + " \"f1Score\": 0.7488228,\n", + " \"falsePositiveRate\": 0.0023712006,\n", + " \"truePositiveCount\": \"2783\",\n", + " \"falsePositiveCount\": \"11\",\n", + " \"falseNegativeCount\": \"1856\",\n", + " \"trueNegativeCount\": \"4628\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.96897614,\n", + " \"recall\": 0.59042895,\n", + " \"precision\": 0.9963623,\n", + " \"f1Score\": 0.74147266,\n", + " \"falsePositiveRate\": 0.002155637,\n", + " \"truePositiveCount\": \"2739\",\n", + " \"falsePositiveCount\": \"10\",\n", + " \"falseNegativeCount\": \"1900\",\n", + " \"trueNegativeCount\": \"4629\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.97092766,\n", + " \"recall\": 0.5807286,\n", + " \"precision\": 0.99667037,\n", + " \"f1Score\": 0.73385996,\n", + " \"falsePositiveRate\": 0.0019400733,\n", + " \"truePositiveCount\": \"2694\",\n", + " \"falsePositiveCount\": \"9\",\n", + " \"falseNegativeCount\": \"1945\",\n", + " \"trueNegativeCount\": \"4630\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9727514,\n", + " \"recall\": 0.5699504,\n", + " \"precision\": 0.9969834,\n", + " \"f1Score\": 0.7252777,\n", + " \"falsePositiveRate\": 0.0017245096,\n", + " \"truePositiveCount\": \"2644\",\n", + " \"falsePositiveCount\": \"8\",\n", + " \"falseNegativeCount\": \"1995\",\n", + " \"trueNegativeCount\": \"4631\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9747563,\n", + " \"recall\": 0.56046563,\n", + " \"precision\": 0.99731493,\n", + " \"f1Score\": 0.7176373,\n", + " \"falsePositiveRate\": 0.001508946,\n", + " \"truePositiveCount\": \"2600\",\n", + " \"falsePositiveCount\": \"7\",\n", + " \"falseNegativeCount\": \"2039\",\n", + " \"trueNegativeCount\": \"4632\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9766838,\n", + " \"recall\": 0.549903,\n", + " \"precision\": 0.9976535,\n", + " \"f1Score\": 0.709005,\n", + " \"falsePositiveRate\": 0.0012933821,\n", + " \"truePositiveCount\": \"2551\",\n", + " \"falsePositiveCount\": \"6\",\n", + " \"falseNegativeCount\": \"2088\",\n", + " \"trueNegativeCount\": \"4633\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.97776866,\n", + " \"recall\": 0.54063374,\n", + " \"precision\": 0.99801034,\n", + " \"f1Score\": 0.7013423,\n", + " \"falsePositiveRate\": 0.0010778185,\n", + " \"truePositiveCount\": \"2508\",\n", + " \"falsePositiveCount\": \"5\",\n", + " \"falseNegativeCount\": \"2131\",\n", + " \"trueNegativeCount\": \"4634\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9792647,\n", + " \"recall\": 0.53050226,\n", + " \"precision\": 0.9979724,\n", + " \"f1Score\": 0.6927516,\n", + " \"falsePositiveRate\": 0.0010778185,\n", + " \"truePositiveCount\": \"2461\",\n", + " \"falsePositiveCount\": \"5\",\n", + " \"falseNegativeCount\": \"2178\",\n", + " \"trueNegativeCount\": \"4634\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9804138,\n", + " \"recall\": 0.5203708,\n", + " \"precision\": 0.99834573,\n", + " \"f1Score\": 0.6841434,\n", + " \"falsePositiveRate\": 0.0008622548,\n", + " \"truePositiveCount\": \"2414\",\n", + " \"falsePositiveCount\": \"4\",\n", + " \"falseNegativeCount\": \"2225\",\n", + " \"trueNegativeCount\": \"4635\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9814273,\n", + " \"recall\": 0.51045483,\n", + " \"precision\": 0.9991561,\n", + " \"f1Score\": 0.6757027,\n", + " \"falsePositiveRate\": 0.0004311274,\n", + " \"truePositiveCount\": \"2368\",\n", + " \"falsePositiveCount\": \"2\",\n", + " \"falseNegativeCount\": \"2271\",\n", + " \"trueNegativeCount\": \"4637\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98243207,\n", + " \"recall\": 0.5005389,\n", + " \"precision\": 0.9991394,\n", + " \"f1Score\": 0.6669539,\n", + " \"falsePositiveRate\": 0.0004311274,\n", + " \"truePositiveCount\": \"2322\",\n", + " \"falsePositiveCount\": \"2\",\n", + " \"falseNegativeCount\": \"2317\",\n", + " \"trueNegativeCount\": \"4637\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9835001,\n", + " \"recall\": 0.49019185,\n", + " \"precision\": 0.9995604,\n", + " \"f1Score\": 0.6577958,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"2274\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2365\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9844133,\n", + " \"recall\": 0.4804915,\n", + " \"precision\": 0.9995516,\n", + " \"f1Score\": 0.6490028,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"2229\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2410\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98528725,\n", + " \"recall\": 0.47079113,\n", + " \"precision\": 0.99954236,\n", + " \"f1Score\": 0.6400938,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"2184\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2455\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.98630506,\n", + " \"recall\": 0.46001294,\n", + " \"precision\": 0.9995316,\n", + " \"f1Score\": 0.6300561,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"2134\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2505\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9871925,\n", + " \"recall\": 0.44988143,\n", + " \"precision\": 0.9995211,\n", + " \"f1Score\": 0.6204846,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"2087\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2552\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9877509,\n", + " \"recall\": 0.44082776,\n", + " \"precision\": 0.99951124,\n", + " \"f1Score\": 0.6118175,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"2045\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2594\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9885219,\n", + " \"recall\": 0.43048072,\n", + " \"precision\": 0.9994995,\n", + " \"f1Score\": 0.6017779,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"1997\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2642\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9892192,\n", + " \"recall\": 0.4199181,\n", + " \"precision\": 0.9994869,\n", + " \"f1Score\": 0.5913783,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"1948\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2691\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9897657,\n", + " \"recall\": 0.41000214,\n", + " \"precision\": 0.9994745,\n", + " \"f1Score\": 0.5814735,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"1902\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2737\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9902156,\n", + " \"recall\": 0.40051734,\n", + " \"precision\": 0.99946207,\n", + " \"f1Score\": 0.57186824,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"1858\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2781\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.990671,\n", + " \"recall\": 0.3901703,\n", + " \"precision\": 0.9994478,\n", + " \"f1Score\": 0.5612403,\n", + " \"falsePositiveRate\": 0.0002155637,\n", + " \"truePositiveCount\": \"1810\",\n", + " \"falsePositiveCount\": \"1\",\n", + " \"falseNegativeCount\": \"2829\",\n", + " \"trueNegativeCount\": \"4638\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99106044,\n", + " \"recall\": 0.38025436,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.5509917,\n", + " \"truePositiveCount\": \"1764\",\n", + " \"falseNegativeCount\": \"2875\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99142385,\n", + " \"recall\": 0.36990732,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.5400472,\n", + " \"truePositiveCount\": \"1716\",\n", + " \"falseNegativeCount\": \"2923\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.991821,\n", + " \"recall\": 0.35999137,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.52940243,\n", + " \"truePositiveCount\": \"1670\",\n", + " \"falseNegativeCount\": \"2969\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99219143,\n", + " \"recall\": 0.35007545,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.5186013,\n", + " \"truePositiveCount\": \"1624\",\n", + " \"falseNegativeCount\": \"3015\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99265563,\n", + " \"recall\": 0.3401595,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.50764036,\n", + " \"truePositiveCount\": \"1578\",\n", + " \"falseNegativeCount\": \"3061\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9930375,\n", + " \"recall\": 0.33045915,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.49675956,\n", + " \"truePositiveCount\": \"1533\",\n", + " \"falseNegativeCount\": \"3106\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9932301,\n", + " \"recall\": 0.32054323,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.48547176,\n", + " \"truePositiveCount\": \"1487\",\n", + " \"falseNegativeCount\": \"3152\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99349546,\n", + " \"recall\": 0.31041172,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.47376212,\n", + " \"truePositiveCount\": \"1440\",\n", + " \"falseNegativeCount\": \"3199\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.993766,\n", + " \"recall\": 0.30006468,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.461615,\n", + " \"truePositiveCount\": \"1392\",\n", + " \"falseNegativeCount\": \"3247\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9940096,\n", + " \"recall\": 0.29014874,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.44979113,\n", + " \"truePositiveCount\": \"1346\",\n", + " \"falseNegativeCount\": \"3293\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9942971,\n", + " \"recall\": 0.27980167,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.4372579,\n", + " \"truePositiveCount\": \"1298\",\n", + " \"falseNegativeCount\": \"3341\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9945464,\n", + " \"recall\": 0.26988575,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.42505518,\n", + " \"truePositiveCount\": \"1252\",\n", + " \"falseNegativeCount\": \"3387\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.994734,\n", + " \"recall\": 0.2601854,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.41293192,\n", + " \"truePositiveCount\": \"1207\",\n", + " \"falseNegativeCount\": \"3432\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9949861,\n", + " \"recall\": 0.25026944,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.40034482,\n", + " \"truePositiveCount\": \"1161\",\n", + " \"falseNegativeCount\": \"3478\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99523896,\n", + " \"recall\": 0.2399224,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.38699582,\n", + " \"truePositiveCount\": \"1113\",\n", + " \"falseNegativeCount\": \"3526\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99545944,\n", + " \"recall\": 0.23043759,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.37456203,\n", + " \"truePositiveCount\": \"1069\",\n", + " \"falseNegativeCount\": \"3570\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9956529,\n", + " \"recall\": 0.22009054,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.36077738,\n", + " \"truePositiveCount\": \"1021\",\n", + " \"falseNegativeCount\": \"3618\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9958653,\n", + " \"recall\": 0.21039017,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.34764025,\n", + " \"truePositiveCount\": \"976\",\n", + " \"falseNegativeCount\": \"3663\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99604857,\n", + " \"recall\": 0.20025867,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.33369252,\n", + " \"truePositiveCount\": \"929\",\n", + " \"falseNegativeCount\": \"3710\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99617386,\n", + " \"recall\": 0.19012718,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.31950733,\n", + " \"truePositiveCount\": \"882\",\n", + " \"falseNegativeCount\": \"3757\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9963188,\n", + " \"recall\": 0.18042682,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.3056976,\n", + " \"truePositiveCount\": \"837\",\n", + " \"falseNegativeCount\": \"3802\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99644965,\n", + " \"recall\": 0.17029533,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.29102966,\n", + " \"truePositiveCount\": \"790\",\n", + " \"falseNegativeCount\": \"3849\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99657893,\n", + " \"recall\": 0.16059496,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.27674592,\n", + " \"truePositiveCount\": \"745\",\n", + " \"falseNegativeCount\": \"3894\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99672806,\n", + " \"recall\": 0.1502479,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.2612444,\n", + " \"truePositiveCount\": \"697\",\n", + " \"falseNegativeCount\": \"3942\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9968369,\n", + " \"recall\": 0.14033197,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.24612476,\n", + " \"truePositiveCount\": \"651\",\n", + " \"falseNegativeCount\": \"3988\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9969655,\n", + " \"recall\": 0.13041604,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.23073989,\n", + " \"truePositiveCount\": \"605\",\n", + " \"falseNegativeCount\": \"4034\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9970963,\n", + " \"recall\": 0.12028454,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.21473928,\n", + " \"truePositiveCount\": \"558\",\n", + " \"falseNegativeCount\": \"4081\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9972023,\n", + " \"recall\": 0.110799745,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.19949543,\n", + " \"truePositiveCount\": \"514\",\n", + " \"falseNegativeCount\": \"4125\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9973033,\n", + " \"recall\": 0.100668244,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.18292205,\n", + " \"truePositiveCount\": \"467\",\n", + " \"falseNegativeCount\": \"4172\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99741936,\n", + " \"recall\": 0.09053675,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.16604072,\n", + " \"truePositiveCount\": \"420\",\n", + " \"falseNegativeCount\": \"4219\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9974875,\n", + " \"recall\": 0.080620825,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.14921205,\n", + " \"truePositiveCount\": \"374\",\n", + " \"falseNegativeCount\": \"4265\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9976044,\n", + " \"recall\": 0.070273764,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.13131924,\n", + " \"truePositiveCount\": \"326\",\n", + " \"falseNegativeCount\": \"4313\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99769753,\n", + " \"recall\": 0.06014227,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.11346076,\n", + " \"truePositiveCount\": \"279\",\n", + " \"falseNegativeCount\": \"4360\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99778664,\n", + " \"recall\": 0.049795214,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.09486653,\n", + " \"truePositiveCount\": \"231\",\n", + " \"falseNegativeCount\": \"4408\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9978742,\n", + " \"recall\": 0.040525977,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.07789517,\n", + " \"truePositiveCount\": \"188\",\n", + " \"falseNegativeCount\": \"4451\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.997992,\n", + " \"recall\": 0.030394481,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.058995817,\n", + " \"truePositiveCount\": \"141\",\n", + " \"falseNegativeCount\": \"4498\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9980989,\n", + " \"recall\": 0.020262988,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.03972111,\n", + " \"truePositiveCount\": \"94\",\n", + " \"falseNegativeCount\": \"4545\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99821925,\n", + " \"recall\": 0.010347057,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.020482184,\n", + " \"truePositiveCount\": \"48\",\n", + " \"falseNegativeCount\": \"4591\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9987167,\n", + " \"precision\": \"NaN\",\n", + " \"f1Score\": \"NaN\",\n", + " \"falseNegativeCount\": \"4639\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 1.0,\n", + " \"precision\": \"NaN\",\n", + " \"f1Score\": \"NaN\",\n", + " \"falseNegativeCount\": \"4639\",\n", + " \"trueNegativeCount\": \"4639\",\n", + " \"positionThreshold\": 1\n", + " }\n", + " ],\n", + " \"confusionMatrix\": {\n", + " \"row\": [\n", + " {\n", + " \"exampleCount\": [\n", + " 3941,\n", + " 147\n", + " ]\n", + " },\n", + " {\n", + " \"exampleCount\": [\n", + " 293,\n", + " 258\n", + " ]\n", + " }\n", + " ],\n", + " \"displayName\": [\n", + " \"1\",\n", + " \"2\"\n", + " ]\n", + " },\n", + " \"auRoc\": 0.94022393,\n", + " \"logLoss\": 0.20077285\n", + " }\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "7y2NlzuBU7jB" + }, + "source": [ + "## Make batch predictions" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "P4U3D4suU7jB" + }, + "source": [ + "### Make a batch prediction file\r\n", + "\r\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "oX4J0k7HU7jE", + "outputId": "118c7235-cbf7-4e31-ed7e-c4f21f868025" + }, + "outputs": [], + "source": [ + "! gsutil cat $IMPORT_FILE | head -n 1 > tmp.csv\n", + "! gsutil cat $IMPORT_FILE | tail -n 10 >> tmp.csv\n", + "\n", + "! cut -d, -f1-16 tmp.csv > batch.csv\n", + "\n", + "gcs_input_uri = \"gs://\" + BUCKET_NAME + \"/batch.csv\"\n", + "\n", + "! gsutil cp batch.csv $gcs_input_uri\n", + "! gsutil cat $gcs_input_uri\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "Age,Job,MaritalStatus,Education,Default,Balance,Housing,Loan,Contact,Day,Month,Duration,Campaign,PDays,Previous,POutcome\n", + "53,management,married,tertiary,no,583,no,no,cellular,17,nov,226,1,184,4,success\n", + "34,admin.,single,secondary,no,557,no,no,cellular,17,nov,224,1,-1,0,unknown\n", + "23,student,single,tertiary,no,113,no,no,cellular,17,nov,266,1,-1,0,unknown\n", + "73,retired,married,secondary,no,2850,no,no,cellular,17,nov,300,1,40,8,failure\n", + "25,technician,single,secondary,no,505,no,yes,cellular,17,nov,386,2,-1,0,unknown\n", + "51,technician,married,tertiary,no,825,no,no,cellular,17,nov,977,3,-1,0,unknown\n", + "71,retired,divorced,primary,no,1729,no,no,cellular,17,nov,456,2,-1,0,unknown\n", + "72,retired,married,secondary,no,5715,no,no,cellular,17,nov,1127,5,184,3,success\n", + "57,blue-collar,married,secondary,no,668,no,no,telephone,17,nov,508,4,-1,0,unknown\n", + "37,entrepreneur,married,secondary,no,2971,no,no,cellular,17,nov,361,2,188,11,other\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_models_batchpredict:migration,old" + }, + "source": [ + "### [projects.locations.models.batchPredict](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models/batchPredict)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "DA_NyCpsU7jF" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "RDhtOV60U7jG", + "outputId": "bc185bec-5fde-49b9-dc04-5b764afe1925" + }, + "outputs": [], + "source": [ + "input_config = {\n", + " \"gcs_source\": {\n", + " \"input_uris\": [gcs_input_uri]\n", + " }\n", + "}\n", + "\n", + "output_config = {\n", + " \"gcs_destination\": {\n", + " \"output_uri_prefix\": \"gs://\" + f\"{BUCKET_NAME}/batch_output/\"\n", + " }\n", + "}\n", + "\n", + "batch_prediction = automl.BatchPredictRequest(\n", + " name=model_id,\n", + " input_config=input_config,\n", + " output_config=output_config\n", + ")\n", + "\n", + "print(MessageToJson(\n", + " batch_prediction.__dict__[\"_pb\"])\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/TBL8310014324731019264\",\n", + " \"inputConfig\": {\n", + " \"gcsSource\": {\n", + " \"inputUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210228161105/batch.csv\"\n", + " ]\n", + " }\n", + " },\n", + " \"outputConfig\": {\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210228161105/batch_output/\"\n", + " }\n", + " }\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "DV1xyH62U7jK" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ORWxLnxjU7jL" + }, + "outputs": [], + "source": [ + "request = clients[\"tables\"].batch_predict(\n", + " model_name=model_id, \n", + " gcs_input_uris=[gcs_input_uri],\n", + " gcs_output_uri_prefix=\"gs://\" + f\"{BUCKET_NAME}/batch_output/\"\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "f-2-0ZboU7jL" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "NEa2poQsU7jL", + "outputId": "c1ef9e7b-d5f1-46b2-aa10-d864c949e7f4" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "fnYRCde3xXie", + "outputId": "7800b730-ccfb-42f6-b8a9-6a3531e830c7" + }, + "outputs": [], + "source": [ + "destination_uri = batch_prediction.output_config.gcs_destination.output_uri_prefix[:-1]\n", + "\n", + "! gsutil ls $destination_uri/*\n", + "! gsutil cat $destination_uri/prediction*/*.csv\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210228161105/batch_output/prediction-bank_20210228161105-2021-02-28T17:28:50.618755Z/errors_1.csv\n", + "gs://migration-ucaip-trainingaip-20210228161105/batch_output/prediction-bank_20210228161105-2021-02-28T17:28:50.618755Z/tables_1.csv\n", + "Age,Job,MaritalStatus,Education,Default,Balance,Housing,Loan,Contact,Day,Month,Duration,Campaign,PDays,Previous,POutcome,errors_Deposit\n", + "Age,Job,MaritalStatus,Education,Default,Balance,Housing,Loan,Contact,Day,Month,Duration,Campaign,PDays,Previous,POutcome,Deposit_1_score,Deposit_2_score\n", + "73,retired,married,secondary,no,2850,no,no,cellular,17,nov,300,1,40,8,failure,0.55540502071380615,0.44459491968154907\n", + "72,retired,married,secondary,no,5715,no,no,cellular,17,nov,1127,5,184,3,success,0.29455980658531189,0.7054402232170105\n", + "51,technician,married,tertiary,no,825,no,no,cellular,17,nov,977,3,-1,0,unknown,0.478891521692276,0.52110844850540161\n", + "53,management,married,tertiary,no,583,no,no,cellular,17,nov,226,1,184,4,success,0.36687871813774109,0.63312125205993652\n", + "23,student,single,tertiary,no,113,no,no,cellular,17,nov,266,1,-1,0,unknown,0.62017822265625,0.37982171773910522\n", + "37,entrepreneur,married,secondary,no,2971,no,no,cellular,17,nov,361,2,188,11,other,0.78739255666732788,0.21260741353034973\n", + "25,technician,single,secondary,no,505,no,yes,cellular,17,nov,386,2,-1,0,unknown,0.7512086033821106,0.24879136681556702\n", + "57,blue-collar,married,secondary,no,668,no,no,telephone,17,nov,508,4,-1,0,unknown,0.86376494169235229,0.13623508810997009\n", + "71,retired,divorced,primary,no,1729,no,no,cellular,17,nov,456,2,-1,0,unknown,0.47541293501853943,0.52458703517913818\n", + "34,admin.,single,secondary,no,557,no,no,cellular,17,nov,224,1,-1,0,unknown,0.90285629034042358,0.097143754363059998\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "rCxhyYjUU7jb" + }, + "source": [ + "## Make online predictions" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_models_deploy:migration,old" + }, + "source": [ + "### [projects.locations.models.deploy](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models/deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "oDV0__8yU7jd" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "fI2-sB5EU7je" + }, + "outputs": [], + "source": [ + "request = clients[\"tables\"].deploy_model(\n", + " model_name=model_id\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "GqwHgW7eU7je" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "nZnqUFbZU7jj", + "outputId": "19dbc64f-1e7c-4dc2-a76b-38c9ad9baefb" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_models_predict:migration,old" + }, + "source": [ + "### [projects.locations.models.predict](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models/predict)" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [ + "#### Prepare data item for online prediction" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [ + "INSTANCE = {\n", + " \"Age\": '58', \n", + " \"Job\": \"managment\", \n", + " \"MaritalStatus\": \"married\", \n", + " \"Education\": \"teritary\", \n", + " \"Default\": \"no\",\n", + " \"Balance\": '2143', \n", + " \"Housing\": \"yes\", \n", + " \"Loan\": \"no\", \n", + " \"Contact\": \"unknown\", \n", + " \"Day\": '5', \n", + " \"Month\": \"may\",\n", + " \"Duration\": '261', \n", + " \"Campaign\": '1', \n", + " \"PDays\": '-1', \n", + " \"Previous\": 0, \n", + " \"POutcome\": \"unknown\"\n", + "}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "VSXoe_SjU7jk" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "1Lhh4bCKU7jk", + "outputId": "9cff79af-cc47-4581-fbd7-ffaa5ec89e48" + }, + "outputs": [], + "source": [ + "instances_list = [INSTANCE]\n", + "instances = [ParseDict(s, Value()) for s in instances_list]\n", + "\n", + "print(MessageToJson(\n", + " automl.PredictRequest(\n", + " name=model_id,\n", + " payload={\"row\": {\"values\": instances}} \n", + " ).__dict__[\"_pb\"])\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/TBL8310014324731019264\",\n", + " \"payload\": {\n", + " \"row\": {\n", + " \"values\": [\n", + " {\n", + " \"Duration\": \"261\",\n", + " \"MaritalStatus\": \"married\",\n", + " \"Loan\": \"no\",\n", + " \"Education\": \"teritary\",\n", + " \"Default\": \"no\",\n", + " \"Campaign\": \"1\",\n", + " \"Balance\": \"2143\",\n", + " \"Contact\": \"unknown\",\n", + " \"Previous\": 0.0,\n", + " \"Day\": \"5\",\n", + " \"Housing\": \"yes\",\n", + " \"PDays\": \"-1\",\n", + " \"Job\": \"managment\",\n", + " \"Month\": \"may\",\n", + " \"Age\": \"58\",\n", + " \"POutcome\": \"unknown\"\n", + " }\n", + " ]\n", + " }\n", + " }\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "JBvEfH6BU7jq" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "mzOwF0DYU7jr" + }, + "outputs": [], + "source": [ + "request = clients[\"tables\"].predict(\n", + " model_name=model_id,\n", + " inputs=INSTANCE\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "4hfPdKUrU7jr" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ZL9Nx4gCU7jr", + "outputId": "7361069f-bd89-48c7-85ee-a93fcc89c35a" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"payload\": [\n", + " {\n", + " \"tables\": {\n", + " \"score\": 0.9947612,\n", + " \"value\": \"1\"\n", + " }\n", + " },\n", + " {\n", + " \"tables\": {\n", + " \"score\": 0.005238751,\n", + " \"value\": \"2\"\n", + " }\n", + " }\n", + " ]\n", + "}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "j14fFDBnxXii" + }, + "source": [ + "### [projects.locations.models.undeploy](https://cloud.google.com/automl/docs/reference/rest/v1/projects.locations.models/undeploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wrhN_FT4xXii" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "kR-o0CO7xXii" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].undeploy_model(\n", + " name=model_id\n", + ")\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "JF1qgjOfxXij" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "-VbCkwYyxXij" + }, + "outputs": [], + "source": [ + "result = request.result()\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "SpuhCYjSxXij" + }, + "source": [ + "#### Response\n", + "\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ot9hRqTVxXij", + "outputId": "1d17e1c1-2be9-4f4f-a25b-404ee7465f15" + }, + "outputs": [], + "source": [ + "print(MessageToJson(result))\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "4tj-vFT9xXij" + }, + "source": [ + "*Example output*:\n", + "\n", + "```\n", + "{}\n", + "```\n", + "\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bQ-VVaSxJjkd" + }, + "source": [ + "# Cleaning up\r\n", + "\r\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\r\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\r\n", + "\r\n", + "Otherwise, you can delete the individual resources you created in this tutorial.\r\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "F5cjBuTKU7jx" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_model = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the AutoML fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset:\n", + " clients['automl'].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the AutoML fully qualified identifier for the model\n", + "try:\n", + " if delete_model:\n", + " clients['automl'].delete_model(name=model_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "\n", + "if delete_bucket and 'BUCKET_NAME' in globals():\n", + " ! gsutil rm -r gs://$BUCKET_NAME\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": {}, + "outputs": [], + "source": [] + } + ], + "metadata": { + "colab": { + "collapsed_sections": [ + "J_croPBoJjkK", + "MDUAZaN3JjkL", + "fR9geV9pJjkO", + "hIHTX-pkJjkO", + "lbv411XjJjkP", + "SBxKjsdoU7h0", + "eNQ_vjVFU7h8", + "text_datasets_create:migration,old", + "ozaMOwdNJjkT", + "vHkuSZxkJjkT", + "FLqkZMD2JjkT", + "text_datasets_importdata:migration,old", + "nacQ0lvTU7iU", + "NEE3oQPOU7ia", + "BfEBJzZ_U7ib", + "zYDxpb87U7ig", + "DS-tkN-lU7ig", + "LoBmmTiHU7ii", + "text_models_create:migration,old", + "wX-bPrRTU7io", + "GJu9d2CFU7it", + "yfufMwAEJjkX", + "SjYvkggyU7iy", + "jFRv00YOU7iz", + "i6T0bzuNJjkY", + "Bvhd2ITWU7i7", + "YOHFYa0rU7i8", + "P4U3D4suU7jB", + "text_models_batchpredict:migration,old", + "DA_NyCpsU7jF", + "DV1xyH62U7jK", + "f-2-0ZboU7jL", + "text_models_deploy:migration,old", + "oDV0__8yU7jd", + "GqwHgW7eU7je", + "text_models_predict:migration,old", + "VSXoe_SjU7jk", + "JBvEfH6BU7jq", + "4hfPdKUrU7jr" + ], + "name": "Kulwant [UJ.4 OLD] AutoML Tables Regression Classification.ipynb", + "provenance": [], + "toc_visible": true + }, + "environment": { + "name": "tf2-2-3-gpu.2-3.m55", + "type": "gcloud", + "uri": "gcr.io/deeplearning-platform-release/tf2-2-3-gpu.2-3:m55" + }, + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.7.8" + } + }, + "nbformat": 4, + "nbformat_minor": 4 +} diff --git a/notebooks/community/migration/UJ4 unified AutoML Tables Regression.ipynb b/notebooks/community/migration/UJ4 unified AutoML Tables Regression.ipynb new file mode 100644 index 000000000..cd9758fa0 --- /dev/null +++ b/notebooks/community/migration/UJ4 unified AutoML Tables Regression.ipynb @@ -0,0 +1,2482 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bur1nAUVj2Xq" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_title:migration,automl,icn" + }, + "source": [ + "# Vertex AI AutoML tables regression\r\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "JE-aKjayJjkF" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest (preview) version of Vertex SDK." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "7dxB6-B0JjkG" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-aiplatform --user" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Iddo-8HmJjkH" + }, + "source": [ + "Install the Google *cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "xHvi6EFSJjkH" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-storage" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ktRTgB8DJjkI" + }, + "source": [ + "### Restart the Kernel\n", + "\n", + "Once you've installed the Vertex SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "x1g4mBdlJjkI" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "if not os.getenv(\"AUTORUN\") and False:\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "neMWauLRtmvU" + }, + "source": [ + "## Before you begin\r\n", + "\r\n", + "### GPU run-time\r\n", + "\r\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\r\n", + "\r\n", + "### Set up your GCP project\r\n", + "\r\n", + "**The following steps are required, regardless of your notebook environment.**\r\n", + "\r\n", + "1. [Select or create a GCP project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\r\n", + "\r\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\r\n", + "\r\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\r\n", + "\r\n", + "4. [Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebooks.\r\n", + "\r\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\r\n", + "Cloud SDK uses the right project for all the commands in this notebook.\r\n", + "\r\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "eidueLnCJjkJ" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "FFyaPu7Yj2X0" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "0UfsdLCEJjkJ" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex AI. We recommend when possible, to choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You cannot use a Multi-Regional Storage bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see [Region support for Vertex AI services](https://cloud.google.com/vertex-ai/docs/general/locations)\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "CTBlncfrJjkK" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "KKxStk5bJjkL" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "qRj4gV6eSVeC" + }, + "source": [ + "### Authenticate your GCP account\n", + "\n", + "**If you are using Google Cloud Notebooks**, your environment is already\n", + "authenticated. Skip this step.\n", + "\n", + "*Note: If you are on an Vertex notebook and run the cell, the cell knows to skip executing the authentication steps.*\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "q96G-uqtukhC" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your Google Cloud account. This provides access\n", + "# to your Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Vertex, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this tutorial in a notebook locally, replace the string\n", + " # below with the path to your service account key and run this cell to\n", + " # authenticate your Google Cloud account.\n", + " else:\n", + " %env GOOGLE_APPLICATION_CREDENTIALS your_path_to_credentials.json\n", + "\n", + " # Log in to your account on Google Cloud\n", + " ! gcloud auth login" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "su5Bpxk4JjkL" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "mAwgOP4BJjkM" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n", + " BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "QVRixM4oJjkM" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aqDptKpYJjkN" + }, + "outputs": [], + "source": [ + "! gsutil ls -al gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip" + }, + "source": [ + "#### Import Vertex SDK\n", + "\n", + "Import the Vertex SDK into our Python environment.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "lghaBjtFj2X5" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex AI constants\n", + "\n", + "Setup up the following constants for Vertex AI:\n", + "\n", + "- `API_ENDPOINT`: The Vertex AI API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `API_PREDICT_ENDPOINT`: The Vertex AI API service endpoint for prediction.\n", + "- `PARENT`: The Vertex AI location root path for dataset, model and endpoint resources.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "IhYGcaMMj2X6" + }, + "outputs": [], + "source": [ + "# API Endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex AI location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants:automl" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Next, setup constants unique to AutoML Table classification datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the managed dataset service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the managed dataset service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the managed pipelines service the task (e.g., classification) to train the model for.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "iTjEIqJ4j2X6" + }, + "outputs": [], + "source": [ + "# Tabular Dataset type\n", + "TABLE_SCHEMA = \"google-cloud-aiplatform/schema/dataset/metadata/tables_1.0.0.yaml\"\n", + "# Tabular Labeling type\n", + "IMPORT_SCHEMA_TABLE_CLASSIFICATION = (\n", + " \"gs://google-cloud-aiplatform/schema/dataset/ioformat/table_io_format_1.0.0.yaml\"\n", + ")\n", + "# Tabular Training task\n", + "TRAINING_TABLE_CLASSIFICATION_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_tables_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients" + }, + "source": [ + "## Clients\n", + "\n", + "The Vertex SDK works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the server (Vertex).\n", + "\n", + "You will use several clients in this tutorial, so set them all up upfront.\n", + "\n", + "- Dataset Service for managed datasets.\n", + "- Model Service for managed models.\n", + "- Pipeline Service for training.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving. *Note*: Prediction has a different service endpoint.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "PclAcK_rj2X7" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "clients[\"job\"] = create_job_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "KY5WZL59j2X8" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://cloud-ml-tables-data/bank-marketing.csv\"" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "_6Eun5MoSVfT" + }, + "outputs": [], + "source": [ + "! gsutil cat $IMPORT_FILE | head -n 10" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "Age,Job,MaritalStatus,Education,Default,Balance,Housing,Loan,Contact,Day,Month,Duration,Campaign,PDays,Previous,POutcome,Deposit\n", + "58,management,married,tertiary,no,2143,yes,no,unknown,5,may,261,1,-1,0,unknown,1\n", + "44,technician,single,secondary,no,29,yes,no,unknown,5,may,151,1,-1,0,unknown,1\n", + "33,entrepreneur,married,secondary,no,2,yes,yes,unknown,5,may,76,1,-1,0,unknown,1\n", + "47,blue-collar,married,unknown,no,1506,yes,no,unknown,5,may,92,1,-1,0,unknown,1\n", + "33,unknown,single,unknown,no,1,no,no,unknown,5,may,198,1,-1,0,unknown,1\n", + "35,management,married,tertiary,no,231,yes,no,unknown,5,may,139,1,-1,0,unknown,1\n", + "28,management,single,tertiary,no,447,yes,yes,unknown,5,may,217,1,-1,0,unknown,1\n", + "42,entrepreneur,divorced,tertiary,yes,2,yes,no,unknown,5,may,380,1,-1,0,unknown,1\n", + "58,retired,married,primary,no,121,yes,no,unknown,5,may,50,1,-1,0,unknown,1\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_create_dataset:migration" + }, + "source": [ + "## Create a dataset" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_datasets_create:migration,new" + }, + "source": [ + "### [projects.locations.datasets.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.datasets/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "YNiPLLgtj2X-" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "GCdGmc3bj2X-" + }, + "outputs": [], + "source": [ + "DATA_SCHEMA = TABLE_SCHEMA\n", + "\n", + "metadata = {\n", + " \"input_config\": {\n", + " \"gcs_source\": {\n", + " \"uri\": [IMPORT_FILE],\n", + " }\n", + " }\n", + "}\n", + "\n", + "dataset = {\n", + " \"display_name\": \"bank_\" + TIMESTAMP,\n", + " \"metadata_schema_uri\": \"gs://\" + DATA_SCHEMA,\n", + " \"metadata\": json_format.ParseDict(metadata, Value()),\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateDatasetRequest(\n", + " parent=PARENT,\n", + " dataset=dataset,\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "nHS5ZJE2SVfp" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"dataset\": {\n", + " \"displayName\": \"bank_20210226015209\",\n", + " \"metadataSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/metadata/tables_1.0.0.yaml\",\n", + " \"metadata\": {\n", + " \"input_config\": {\n", + " \"gcs_source\": {\n", + " \"uri\": [\n", + " \"gs://cloud-ml-tables-data/bank-marketing.csv\"\n", + " ]\n", + " }\n", + " }\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "jgSHrqOTj2X-" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "r38jteRRj2X-" + }, + "outputs": [], + "source": [ + "request = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tEqBFRvOj2X_" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pqr16orDj2X_" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "FxhTCGqHSVfx" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/datasets/7748812594797871104\",\n", + " \"displayName\": \"bank_20210226015209\",\n", + " \"metadataSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/metadata/tabular_1.0.0.yaml\",\n", + " \"labels\": {\n", + " \"aiplatform.googleapis.com/dataset_metadata_schema\": \"TABLE\"\n", + " },\n", + " \"metadata\": {\n", + " \"inputConfig\": {\n", + " \"gcsSource\": {\n", + " \"uri\": [\n", + " \"gs://cloud-ml-tables-data/bank-marketing.csv\"\n", + " ]\n", + " }\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "hdLkCw9fSVf0" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_create_and_deploy_model:migration" + }, + "source": [ + "## Train a model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0oqIBOSnJjkW" + }, + "source": [ + "### [projects.locations.trainingPipelines.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.trainingPipelines/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "FPUS2ZXKj2X_" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ddejk5QJj2YA" + }, + "outputs": [], + "source": [ + "TRAINING_SCHEMA = TRAINING_TABLE_CLASSIFICATION_SCHEMA\n", + "\n", + "TRANSFORMATIONS = [\n", + " {\"auto\": {\"column_name\": \"Age\"}},\n", + " {\"auto\": {\"column_name\": \"Job\"}},\n", + " {\"auto\": {\"column_name\": \"MaritalStatus\"}},\n", + " {\"auto\": {\"column_name\": \"Education\"}},\n", + " {\"auto\": {\"column_name\": \"Default\"}},\n", + " {\"auto\": {\"column_name\": \"Balance\"}},\n", + " {\"auto\": {\"column_name\": \"Housing\"}},\n", + " {\"auto\": {\"column_name\": \"Loan\"}},\n", + " {\"auto\": {\"column_name\": \"Contact\"}},\n", + " {\"auto\": {\"column_name\": \"Day\"}},\n", + " {\"auto\": {\"column_name\": \"Month\"}},\n", + " {\"auto\": {\"column_name\": \"Duration\"}},\n", + " {\"auto\": {\"column_name\": \"Campaign\"}},\n", + " {\"auto\": {\"column_name\": \"PDays\"}},\n", + " {\"auto\": {\"column_name\": \"POutcome\"}},\n", + "]\n", + "\n", + "task = Value(\n", + " struct_value=Struct(\n", + " fields={\n", + " \"disable_early_stopping\": Value(bool_value=False),\n", + " \"prediction_type\": Value(string_value=\"regression\"),\n", + " \"target_column\": Value(string_value=\"Deposit\"),\n", + " \"train_budget_milli_node_hours\": Value(number_value=1000),\n", + " \"transformations\": json_format.ParseDict(TRANSFORMATIONS, Value()),\n", + " }\n", + " )\n", + ")\n", + "\n", + "training_pipeline = {\n", + " \"display_name\": \"bank_\" + TIMESTAMP,\n", + " \"input_data_config\": {\n", + " \"dataset_id\": dataset_short_id,\n", + " \"fraction_split\": {\n", + " \"training_fraction\": 0.8,\n", + " \"validation_fraction\": 0.1,\n", + " \"test_fraction\": 0.1,\n", + " },\n", + " },\n", + " \"model_to_upload\": {\n", + " \"display_name\": \"flowers_\" + TIMESTAMP,\n", + " },\n", + " \"training_task_definition\": TRAINING_SCHEMA,\n", + " \"training_task_inputs\": task,\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateTrainingPipelineRequest(\n", + " parent=PARENT,\n", + " training_pipeline=training_pipeline,\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "a81NEJ5sSVf5" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"trainingPipeline\": {\n", + " \"displayName\": \"bank_20210226015209\",\n", + " \"inputDataConfig\": {\n", + " \"datasetId\": \"7748812594797871104\",\n", + " \"fractionSplit\": {\n", + " \"trainingFraction\": 0.8,\n", + " \"validationFraction\": 0.1,\n", + " \"testFraction\": 0.1\n", + " }\n", + " },\n", + " \"trainingTaskDefinition\": \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_tables_1.0.0.yaml\",\n", + " \"trainingTaskInputs\": {\n", + " \"transformations\": [\n", + " {\n", + " \"auto\": {\n", + " \"column_name\": \"Age\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"column_name\": \"Job\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"column_name\": \"MaritalStatus\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"column_name\": \"Education\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"column_name\": \"Default\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"column_name\": \"Balance\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"column_name\": \"Housing\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"column_name\": \"Loan\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"column_name\": \"Contact\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"column_name\": \"Day\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"column_name\": \"Month\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"column_name\": \"Duration\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"column_name\": \"Campaign\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"column_name\": \"PDays\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"column_name\": \"POutcome\"\n", + " }\n", + " }\n", + " ],\n", + " \"prediction_type\": \"regression\",\n", + " \"disable_early_stopping\": false,\n", + " \"train_budget_milli_node_hours\": 1000.0,\n", + " \"target_column\": \"Deposit\"\n", + " },\n", + " \"modelToUpload\": {\n", + " \"displayName\": \"flowers_20210226015209\"\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "1Y3qKNmqj2YA" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "hnO0tFVcj2YA" + }, + "outputs": [], + "source": [ + "request = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "IV19l-Oej2YA" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "w9dBuPEGj2YA" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "YqhSktlTSVf9" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/trainingPipelines/3147717072369221632\",\n", + " \"displayName\": \"bank_20210226015209\",\n", + " \"inputDataConfig\": {\n", + " \"datasetId\": \"7748812594797871104\",\n", + " \"fractionSplit\": {\n", + " \"trainingFraction\": 0.8,\n", + " \"validationFraction\": 0.1,\n", + " \"testFraction\": 0.1\n", + " }\n", + " },\n", + " \"trainingTaskDefinition\": \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_tables_1.0.0.yaml\",\n", + " \"trainingTaskInputs\": {\n", + " \"targetColumn\": \"Deposit\",\n", + " \"trainBudgetMilliNodeHours\": \"1000\",\n", + " \"transformations\": [\n", + " {\n", + " \"auto\": {\n", + " \"columnName\": \"Age\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"columnName\": \"Job\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"columnName\": \"MaritalStatus\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"columnName\": \"Education\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"columnName\": \"Default\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"columnName\": \"Balance\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"columnName\": \"Housing\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"columnName\": \"Loan\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"columnName\": \"Contact\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"columnName\": \"Day\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"columnName\": \"Month\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"columnName\": \"Duration\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"columnName\": \"Campaign\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"columnName\": \"PDays\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"columnName\": \"POutcome\"\n", + " }\n", + " }\n", + " ],\n", + " \"predictionType\": \"regression\"\n", + " },\n", + " \"modelToUpload\": {\n", + " \"displayName\": \"flowers_20210226015209\"\n", + " },\n", + " \"state\": \"PIPELINE_STATE_PENDING\",\n", + " \"createTime\": \"2021-02-26T01:57:51.364312Z\",\n", + " \"updateTime\": \"2021-02-26T01:57:51.364312Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "3MXUQ1LuSVf9" + }, + "outputs": [], + "source": [ + "# The full unique ID for the training pipeline\n", + "training_pipeline_id = request.name\n", + "# The short numeric ID for the training pipeline\n", + "training_pipeline_short_id = training_pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(training_pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "QJsV1xoSJjkW" + }, + "source": [ + "### [projects.locations.trainingPipelines.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.trainingPipelines/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Pdd4J3u0j2YB" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "RT-IcGfsj2YB" + }, + "outputs": [], + "source": [ + "request = clients[\"pipeline\"].get_training_pipeline(name=training_pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "N211vNIJj2YB" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "TKY_2wvsj2YB" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gDU5CGQbSVgP" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/trainingPipelines/3147717072369221632\",\n", + " \"displayName\": \"bank_20210226015209\",\n", + " \"inputDataConfig\": {\n", + " \"datasetId\": \"7748812594797871104\",\n", + " \"fractionSplit\": {\n", + " \"trainingFraction\": 0.8,\n", + " \"validationFraction\": 0.1,\n", + " \"testFraction\": 0.1\n", + " }\n", + " },\n", + " \"trainingTaskDefinition\": \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_tables_1.0.0.yaml\",\n", + " \"trainingTaskInputs\": {\n", + " \"trainBudgetMilliNodeHours\": \"1000\",\n", + " \"transformations\": [\n", + " {\n", + " \"auto\": {\n", + " \"columnName\": \"Age\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"columnName\": \"Job\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"columnName\": \"MaritalStatus\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"columnName\": \"Education\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"columnName\": \"Default\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"columnName\": \"Balance\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"columnName\": \"Housing\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"columnName\": \"Loan\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"columnName\": \"Contact\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"columnName\": \"Day\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"columnName\": \"Month\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"columnName\": \"Duration\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"columnName\": \"Campaign\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"columnName\": \"PDays\"\n", + " }\n", + " },\n", + " {\n", + " \"auto\": {\n", + " \"columnName\": \"POutcome\"\n", + " }\n", + " }\n", + " ],\n", + " \"targetColumn\": \"Deposit\",\n", + " \"predictionType\": \"regression\"\n", + " },\n", + " \"modelToUpload\": {\n", + " \"displayName\": \"flowers_20210226015209\"\n", + " },\n", + " \"state\": \"PIPELINE_STATE_PENDING\",\n", + " \"createTime\": \"2021-02-26T01:57:51.364312Z\",\n", + " \"updateTime\": \"2021-02-26T01:57:51.364312Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "S3R6WFWBSVgQ" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=training_pipeline_id)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_name = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " break\n", + " else:\n", + " model_id = response.model_to_upload.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_evaluate_the_model:migration" + }, + "source": [ + "## Evaluate the model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "BJNzhELXJjkY" + }, + "source": [ + "### [projects.locations.models.evaluations.list](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.models.evaluations/list)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "6wNcEDinj2YC" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Zh6bGAI9j2YC" + }, + "outputs": [], + "source": [ + "request = clients[\"model\"].list_model_evaluations(parent=model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ewprNQq_j2YC" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "AxEY7WFHj2YC" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "model_evaluations = [json.loads(MessageToJson(me.__dict__[\"_pb\"])) for me in request]\n", + "# The evaluation slice\n", + "evaluation_slice = request.model_evaluations[0].name\n", + "\n", + "print(json.dumps(model_evaluations, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "D8ddmURpSVgT" + }, + "source": [ + "*Example output*:\n", + "```\n", + "[\n", + " {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/3936304403996213248/evaluations/6323797633322037836\",\n", + " \"metricsSchemaUri\": \"gs://google-cloud-aiplatform/schema/modelevaluation/regression_metrics_1.0.0.yaml\",\n", + " \"metrics\": {\n", + " \"rSquared\": 0.39799774,\n", + " \"meanAbsolutePercentageError\": 9.791032,\n", + " \"rootMeanSquaredError\": 0.24675915,\n", + " \"rootMeanSquaredLogError\": 0.10022795,\n", + " \"meanAbsoluteError\": 0.12842195\n", + " },\n", + " \"createTime\": \"2021-02-26T03:39:42.254525Z\"\n", + " }\n", + "]\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wr2JuU5nJjka" + }, + "source": [ + "### [projects.locations.models.evaluations.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.models.evaluations/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "PqOnHBrKj2YD" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "20ZIQFkwj2YD" + }, + "outputs": [], + "source": [ + "request = clients[\"model\"].get_model_evaluation(name=evaluation_slice)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "jABbeZQDj2YD" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "A8qKdt2bj2YD" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ZdZOaLg6SVgY" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/3936304403996213248/evaluations/6323797633322037836\",\n", + " \"metricsSchemaUri\": \"gs://google-cloud-aiplatform/schema/modelevaluation/regression_metrics_1.0.0.yaml\",\n", + " \"metrics\": {\n", + " \"meanAbsolutePercentageError\": 9.791032,\n", + " \"rootMeanSquaredLogError\": 0.10022795,\n", + " \"rSquared\": 0.39799774,\n", + " \"meanAbsoluteError\": 0.12842195,\n", + " \"rootMeanSquaredError\": 0.24675915\n", + " },\n", + " \"createTime\": \"2021-02-26T03:39:42.254525Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "opns5WnQj2YD" + }, + "source": [ + "## Make batch predictions" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "RqHaBQODj2YE" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "Let's now make a batch input file, which you store in your local Cloud Storage bucket. The batch input file can be either CSV or JSONL. You will use CSV in this tutorial.ion on." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "6J33oj4zj2YE" + }, + "outputs": [], + "source": [ + "! gsutil cat $IMPORT_FILE | head -n 1 > tmp.csv\n", + "! gsutil cat $IMPORT_FILE | tail -n 10 >> tmp.csv\n", + "\n", + "! cut -d, -f1-16 tmp.csv > batch.csv\n", + "\n", + "gcs_input_uri = \"gs://\" + BUCKET_NAME + \"/test.csv\"\n", + "\n", + "! gsutil cp batch.csv $gcs_input_uri" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "409yPtWkSVga" + }, + "outputs": [], + "source": [ + "! gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "lRG34PmTSVgd" + }, + "source": [ + "*Example output*:\n", + "```\n", + "Age,Job,MaritalStatus,Education,Default,Balance,Housing,Loan,Contact,Day,Month,Duration,Campaign,PDays,Previous,POutcome\n", + "53,management,married,tertiary,no,583,no,no,cellular,17,nov,226,1,184,4,success\n", + "34,admin.,single,secondary,no,557,no,no,cellular,17,nov,224,1,-1,0,unknown\n", + "23,student,single,tertiary,no,113,no,no,cellular,17,nov,266,1,-1,0,unknown\n", + "73,retired,married,secondary,no,2850,no,no,cellular,17,nov,300,1,40,8,failure\n", + "25,technician,single,secondary,no,505,no,yes,cellular,17,nov,386,2,-1,0,unknown\n", + "51,technician,married,tertiary,no,825,no,no,cellular,17,nov,977,3,-1,0,unknown\n", + "71,retired,divorced,primary,no,1729,no,no,cellular,17,nov,456,2,-1,0,unknown\n", + "72,retired,married,secondary,no,5715,no,no,cellular,17,nov,1127,5,184,3,success\n", + "57,blue-collar,married,secondary,no,668,no,no,telephone,17,nov,508,4,-1,0,unknown\n", + "37,entrepreneur,married,secondary,no,2971,no,no,cellular,17,nov,361,2,188,11,other\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "9Fs3ltvZj2YE" + }, + "source": [ + "### [projects.locations.batchPredictionJobs.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.batchPredictionJobs/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ozaMOwdNJjkT" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "PIUXjvd5j2YE" + }, + "outputs": [], + "source": [ + "batch_prediction_job = {\n", + " \"display_name\": \"bank_\" + TIMESTAMP,\n", + " \"model\": model_id,\n", + " \"input_config\": {\n", + " \"instances_format\": \"csv\",\n", + " \"gcs_source\": {\n", + " \"uris\": [gcs_input_uri],\n", + " },\n", + " },\n", + " \"output_config\": {\n", + " \"predictions_format\": \"csv\",\n", + " \"gcs_destination\": {\n", + " \"output_uri_prefix\": \"gs://\" + f\"{BUCKET_NAME}/batch_output/\",\n", + " },\n", + " },\n", + " \"dedicated_resources\": {\n", + " \"machine_spec\": {\n", + " \"machine_type\": \"n1-standard-2\",\n", + " \"accelerator_count\": 0,\n", + " },\n", + " \"starting_replica_count\": 1,\n", + " \"max_replica_count\": 1,\n", + " },\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateBatchPredictionJobRequest(\n", + " parent=PARENT,\n", + " batch_prediction_job=batch_prediction_job,\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ZlIBPI0ySVgg" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"batchPredictionJob\": {\n", + " \"displayName\": \"bank_20210226015209\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/3936304403996213248\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"csv\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210226015209/test.csv\"\n", + " ]\n", + " }\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"csv\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210226015209/batch_output/\"\n", + " }\n", + " },\n", + " \"dedicatedResources\": {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-2\"\n", + " },\n", + " \"startingReplicaCount\": 1,\n", + " \"maxReplicaCount\": 1\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "vHkuSZxkJjkT" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "B3uK5lMej2YF" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].create_batch_prediction_job(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "FLqkZMD2JjkT" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "GuKiQ5e7j2YF" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "_DDEH1DJSVgk" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/batchPredictionJobs/4417450692310990848\",\n", + " \"displayName\": \"bank_20210226015209\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/3936304403996213248\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"csv\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210226015209/test.csv\"\n", + " ]\n", + " }\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"csv\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210226015209/batch_output/\"\n", + " }\n", + " },\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"createTime\": \"2021-02-26T09:35:43.113270Z\",\n", + " \"updateTime\": \"2021-02-26T09:35:43.113270Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "H-4_9jp_SVgl" + }, + "outputs": [], + "source": [ + "# The fully qualified ID for the batch job\n", + "batch_job_id = request.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split(\"/\")[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "EQL9wcWHj2YF" + }, + "source": [ + "### [projects.locations.batchPredictionJobs.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.batchPredictionJobs/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "AXE_K1nHj2YF" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "i6osHKnnj2YG" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].get_batch_prediction_job(name=batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "j0-DnHS0j2YG" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cK5Xz66oj2YG" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "2zM2M3jrSVgo" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/batchPredictionJobs/4417450692310990848\",\n", + " \"displayName\": \"bank_20210226015209\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/3936304403996213248\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"csv\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210226015209/test.csv\"\n", + " ]\n", + " }\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"csv\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210226015209/batch_output/\"\n", + " }\n", + " },\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"createTime\": \"2021-02-26T09:35:43.113270Z\",\n", + " \"updateTime\": \"2021-02-26T09:35:43.113270Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "def get_latest_predictions(gcs_out_dir):\n", + " \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n", + " folders = !gsutil ls $gcs_out_dir\n", + " latest = \"\"\n", + " for folder in folders:\n", + " subfolder = folder.split(\"/\")[-2]\n", + " if subfolder.startswith(\"prediction-\"):\n", + " if subfolder > latest:\n", + " latest = folder[:-1]\n", + " return latest\n", + "\n", + "\n", + "while True:\n", + " response = clients[\"job\"].get_batch_prediction_job(name=batch_job_id)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", response.state)\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " folder = get_latest_predictions(\n", + " response.output_config.gcs_destination.output_uri_prefix\n", + " )\n", + " ! gsutil ls $folder/prediction*\n", + "\n", + " ! gsutil cat $folder/prediction*\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "TcSzUOeMSVgp" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210226015209/batch_output/prediction-flowers_20210226015209-2021-02-26T09:35:43.034287Z/predictions_1.csv\n", + "Age,Balance,Campaign,Contact,Day,Default,Duration,Education,Housing,Job,Loan,MaritalStatus,Month,PDays,POutcome,Previous,predicted_Deposit\n", + "72,5715,5,cellular,17,no,1127,secondary,no,retired,no,married,nov,184,success,3,1.6232702732086182\n", + "23,113,1,cellular,17,no,266,tertiary,no,student,no,single,nov,-1,unknown,0,1.3257474899291992\n", + "34,557,1,cellular,17,no,224,secondary,no,admin.,no,single,nov,-1,unknown,0,1.0801490545272827\n", + "25,505,2,cellular,17,no,386,secondary,no,technician,yes,single,nov,-1,unknown,0,1.2516863346099854\n", + "73,2850,1,cellular,17,no,300,secondary,no,retired,no,married,nov,40,failure,8,1.5064295530319214\n", + "37,2971,2,cellular,17,no,361,secondary,no,entrepreneur,no,married,nov,188,other,11,1.1924527883529663\n", + "57,668,4,telephone,17,no,508,secondary,no,blue-collar,no,married,nov,-1,unknown,0,1.1636843681335449\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "HYO2mVhGj2YG" + }, + "source": [ + "## Make online predictions" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "COwVZtxhJjkW" + }, + "source": [ + "### [projects.locations.endpoints.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dFkaAtoQj2YH" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Gg16KcNsj2YH" + }, + "outputs": [], + "source": [ + "endpoint = {\"display_name\": \"bank_\" + TIMESTAMP}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateEndpointRequest(\n", + " parent=PARENT,\n", + " endpoint=endpoint,\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "72sCWPDUSVgr" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"endpoint\": {\n", + " \"displayName\": \"bank_20210226015209\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "zvwqgIF6j2YH" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "DrlxXQmfj2YH" + }, + "outputs": [], + "source": [ + "request = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tBX51eh3j2YH" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "UpDCfsaAj2YH" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gQOxjES2SVgt" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/endpoints/6899338707271155712\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "oUKwpVp6SVgu" + }, + "outputs": [], + "source": [ + "# The fully qualified ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gpnVJS3AJjkW" + }, + "source": [ + "### [projects.locations.endpoints.deployModel](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/deployModel)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ZhSFjr7dj2YI" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "9WH1Rhh6j2YI" + }, + "outputs": [], + "source": [ + "deployed_model = {\n", + " \"model\": model_id,\n", + " \"display_name\": \"bank_\" + TIMESTAMP,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": 1,\n", + " \"machine_spec\": {\"machine_type\": \"n1-standard-2\"},\n", + " },\n", + "}\n", + "\n", + "traffic_split = {\"0\": 100}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.DeployModelRequest(\n", + " endpoint=endpoint_id,\n", + " deployed_model=deployed_model,\n", + " traffic_split=traffic_split,\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "nPn4WRMHSVgv" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"endpoint\": \"projects/116273516712/locations/us-central1/endpoints/6899338707271155712\",\n", + " \"deployedModel\": {\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/3936304403996213248\",\n", + " \"displayName\": \"bank_20210226015209\",\n", + " \"dedicatedResources\": {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-2\"\n", + " },\n", + " \"minReplicaCount\": 1\n", + " }\n", + " },\n", + " \"trafficSplit\": {\n", + " \"0\": 100\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "g5vJaOgOj2YI" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "x-cdr_Y_j2YI" + }, + "outputs": [], + "source": [ + "request = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint_id, deployed_model=deployed_model, traffic_split=traffic_split\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "zPkahTYxj2YI" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "RV7K38Tgj2YJ" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "v6zgV9_ISVgy" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"deployedModel\": {\n", + " \"id\": \"7646795507926302720\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "D2nPnJFXSVgz" + }, + "outputs": [], + "source": [ + "# The numeric ID for the deploy model\n", + "deploy_model_id = result.deployed_model.id\n", + "\n", + "print(deploy_model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "yFL5-rTvj2YJ" + }, + "source": [ + "### [projects.locations.endpoints.predict](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/predict)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "6665b445b824" + }, + "source": [ + "### Prepare data item for online prediction" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "445a132a6fcf" + }, + "outputs": [], + "source": [ + "INSTANCE = {\n", + " \"Age\": \"58\",\n", + " \"Job\": \"managment\",\n", + " \"MaritalStatus\": \"married\",\n", + " \"Education\": \"teritary\",\n", + " \"Default\": \"no\",\n", + " \"Balance\": \"2143\",\n", + " \"Housing\": \"yes\",\n", + " \"Loan\": \"no\",\n", + " \"Contact\": \"unknown\",\n", + " \"Day\": \"5\",\n", + " \"Month\": \"may\",\n", + " \"Duration\": \"261\",\n", + " \"Campaign\": \"1\",\n", + " \"PDays\": \"-1\",\n", + " \"Previous\": 0,\n", + " \"POutcome\": \"unknown\",\n", + "}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "xsz7Bbtcj2YJ" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Yb0-Lf5Rj2YJ" + }, + "outputs": [], + "source": [ + "instances_list = [INSTANCE]\n", + "instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + "request = aip.PredictRequest(\n", + " endpoint=endpoint_id,\n", + ")\n", + "request.instances.append(instances)\n", + "\n", + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "KQ1gXsUCSVg2" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"endpoint\": \"projects/116273516712/locations/us-central1/endpoints/6899338707271155712\",\n", + " \"instances\": [\n", + " [\n", + " {\n", + " \"Education\": \"teritary\",\n", + " \"MaritalStatus\": \"married\",\n", + " \"Balance\": \"2143\",\n", + " \"Contact\": \"unknown\",\n", + " \"Housing\": \"yes\",\n", + " \"Previous\": 0.0,\n", + " \"Loan\": \"no\",\n", + " \"Duration\": \"261\",\n", + " \"Default\": \"no\",\n", + " \"Day\": \"5\",\n", + " \"POutcome\": \"unknown\",\n", + " \"Age\": \"58\",\n", + " \"Month\": \"may\",\n", + " \"PDays\": \"-1\",\n", + " \"Campaign\": \"1\",\n", + " \"Job\": \"managment\"\n", + " }\n", + " ]\n", + " ]\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "PJl0qAT9j2YJ" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Y9nGPUf-j2YK" + }, + "outputs": [], + "source": [ + "request = clients[\"prediction\"].predict(endpoint=endpoint_id, instances=instances)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "NgHoeLP1j2YK" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "y_DTfVKTj2YK" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pycRDGsnSVg5" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"predictions\": [\n", + " {\n", + " \"upper_bound\": 1.685426712036133,\n", + " \"value\": 1.007092595100403,\n", + " \"lower_bound\": 0.06719603389501572\n", + " }\n", + " ],\n", + " \"deployedModelId\": \"7646795507926302720\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new" + }, + "source": [ + "### [projects.locations.endpoints.undeployModel](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/undeployModel)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "call:migration" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_undeploymodel:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint_id, deployed_model_id=deploy_model_id, traffic_split={}\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "response:migration" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,response" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "RiYgMO0LSVg7" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bQ-VVaSxJjkd" + }, + "source": [ + "# Cleaning up\r\n", + "\r\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\r\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\r\n", + "\r\n", + "Otherwise, you can delete the individual resources you created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "hqvunhZkj2YK" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_pipeline = True\n", + "delete_batchjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex AI fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset:\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex AI fully qualified identifier for the model\n", + "try:\n", + " if delete_model:\n", + " clients[\"model\"].delete_model(name=model_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex AI fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint:\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex AI fully qualified identifier for the training pipeline\n", + "try:\n", + " if delete_pipeline:\n", + " clients[\"pipeline\"].delete_training_pipeline(name=training_pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex AI fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob:\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r gs://$BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "collapsed_sections": [ + "qRj4gV6eSVeC", + "bucket:batch_prediction", + "setup_vars", + "import_aip", + "aip_constants", + "automl_constants:automl", + "endpoints_undeploymodel:migration,new", + "call:migration", + "response:migration" + ], + "name": "UJ4 unified AutoML Tables Regression.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/migration/UJ5 legacy AutoML Vision Images Object Detection.ipynb b/notebooks/community/migration/UJ5 legacy AutoML Vision Images Object Detection.ipynb new file mode 100644 index 000000000..72cfc70c5 --- /dev/null +++ b/notebooks/community/migration/UJ5 legacy AutoML Vision Images Object Detection.ipynb @@ -0,0 +1,1839 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "9nuf2Zwjyaap" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_title:migration,automl,icn" + }, + "source": [ + "# AutoML SDK: AutoML image object detection model\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "JE-aKjayJjkF" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest (preview) version of AutoML SDK." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "7dxB6-B0JjkG" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-automl --user" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Iddo-8HmJjkH" + }, + "source": [ + "Install the Google *cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "xHvi6EFSJjkH" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-storage" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ktRTgB8DJjkI" + }, + "source": [ + "### Restart the Kernel\n", + "\n", + "Once you've installed the AutoML SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "x1g4mBdlJjkI" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "if not os.getenv(\"AUTORUN\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "UvmMLVFSzK1P" + }, + "source": [ + "## Before you begin\r\n", + "\r\n", + "### GPU run-time\r\n", + "\r\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\r\n", + "\r\n", + "### Set up your GCP project\r\n", + "\r\n", + "**The following steps are required, regardless of your notebook environment.**\r\n", + "\r\n", + "1. [Select or create a GCP project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\r\n", + "\r\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\r\n", + "\r\n", + "3. [Enable the AutoML APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\r\n", + "\r\n", + "4. [Google Cloud SDK](https://cloud.google.com/sdk) is already installed in AutoML Notebooks.\r\n", + "\r\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\r\n", + "Cloud SDK uses the right project for all the commands in this notebook.\r\n", + "\r\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "hceMthWGJjkI" + }, + "source": [ + "#### Project ID\n", + "\n", + "**If you don't know your project ID**, try to get your project ID using `gcloud` command by executing the second cell below." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "jT0fijejJjkK" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for AutoML. We recommend when possible, to choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You cannot use a Multi-Regional Storage bucket for training with AutoML. Not all regions provide support for all AutoML services. For the latest support per region, see [Region support for AutoML services]()\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "CTBlncfrJjkK" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "J_croPBoJjkK" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "KKxStk5bJjkL" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "kScwVlYG4K3g" + }, + "source": [ + "### Authenticate your GCP account\r\n", + "\r\n", + "**If you are using AutoML Notebooks**, your environment is already\r\n", + "authenticated. Skip this step.\r\n", + "\r\n", + "*Note: If you are on an AutoML notebook and run the cell, the cell knows to skip executing the authentication steps.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "chybg3Ap_i_2" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your Google Cloud account. This provides access\n", + "# to your Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Vertex, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this tutorial in a notebook locally, replace the string\n", + " # below with the path to your service account key and run this cell to\n", + " # authenticate your Google Cloud account.\n", + " else:\n", + " %env GOOGLE_APPLICATION_CREDENTIALS your_path_to_credentials.json\n", + "\n", + " # Log in to your account on Google Cloud\n", + " ! gcloud auth login" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "MDUAZaN3JjkL" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n", + " BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "yU6VuylGJjkM" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "QVRixM4oJjkM" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Mi-Pdh6QJjkN" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aqDptKpYJjkN" + }, + "outputs": [], + "source": [ + "! gsutil ls -al gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "fR9geV9pJjkO" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "hIHTX-pkJjkO" + }, + "source": [ + "#### Import AutoML SDK\n", + "\n", + "Import the AutoM SDK into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "qrSkIKVgJjkO" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "import time\n", + "\n", + "from google.cloud import automl\n", + "from google.protobuf.json_format import MessageToJson" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "lbv411XjJjkP" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Setup up the following constants for AutoML:\n", + "\n", + "- `PARENT`: The AutoM location root path for dataset, model and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Gx8Eos8PJjkP" + }, + "outputs": [], + "source": [ + "# AutoM location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "i8zg7LVoyaa2" + }, + "source": [ + "## Clients\n", + "\n", + "The AutoML SDK works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the server (AutoML).\n", + "\n", + "You will use several clients in this tutorial, so set them all up upfront.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "c3sR3OxmxXiR" + }, + "outputs": [], + "source": [ + "def automl_client():\n", + " return automl.AutoMlClient()\n", + "\n", + "\n", + "def prediction_client():\n", + " return automl.PredictionServiceClient()\n", + "\n", + "\n", + "def operations_client():\n", + " return automl.AutoMlClient()._transport.operations_client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"automl\"] = automl_client()\n", + "clients[\"prediction\"] = prediction_client()\n", + "clients[\"operations\"] = operations_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:flowers,csv,icn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://cloud-ml-data/img/openimage/csv/salads_ml_use.csv\"" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "2do-AnOHxXiS" + }, + "outputs": [], + "source": [ + "%%capture\n", + "! gsutil cp -r gs://cloud-ml-data/img/openimage/ gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "05fa67c5e593" + }, + "outputs": [], + "source": [ + "! gsutil ls gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "a49c04d37ff9" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "all_files_csv = ! gsutil cat $IMPORT_FILE\n", + "all_files_csv = [l.replace(\"cloud-ml-data/img\", BUCKET_NAME) for l in all_files_csv]\n", + "\n", + "IMPORT_FILE = \"gs://\" + BUCKET_NAME + \"/openimage/salads_ml_use.csv\"\n", + "with tf.io.gfile.GFile(IMPORT_FILE, \"w\") as f:\n", + " for l in all_files_csv:\n", + " f.write(l + \"\\n\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "7e329544a2db" + }, + "outputs": [], + "source": [ + "! gsutil cat $IMPORT_FILE | head -n 10" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "TEST,gs://migration-ucaip-trainingaip-20210301091741/openimage/103/279324025_3e74a32a84_o.jpg,Baked Goods,0.005743,0.084985,,,0.567511,0.735736,,\n", + "TEST,gs://migration-ucaip-trainingaip-20210301091741/openimage/103/279324025_3e74a32a84_o.jpg,Salad,0.402759,0.310473,,,1.000000,0.982695,,\n", + "TEST,gs://migration-ucaip-trainingaip-20210301091741/openimage/1064/3167707458_7b2eebed9e_o.jpg,Cheese,0.000000,0.000000,,,0.054865,0.480665,,\n", + "TEST,gs://migration-ucaip-trainingaip-20210301091741/openimage/1064/3167707458_7b2eebed9e_o.jpg,Cheese,0.041131,0.401678,,,0.318230,0.785916,,\n", + "TEST,gs://migration-ucaip-trainingaip-20210301091741/openimage/1064/3167707458_7b2eebed9e_o.jpg,Cheese,0.116263,0.065161,,,0.451528,0.286489,,\n", + "TEST,gs://migration-ucaip-trainingaip-20210301091741/openimage/1064/3167707458_7b2eebed9e_o.jpg,Cheese,0.557359,0.411551,,,0.988760,0.731613,,\n", + "TEST,gs://migration-ucaip-trainingaip-20210301091741/openimage/1064/3167707458_7b2eebed9e_o.jpg,Cheese,0.562206,0.059401,,,0.876467,0.260982,,\n", + "TEST,gs://migration-ucaip-trainingaip-20210301091741/openimage/1064/3167707458_7b2eebed9e_o.jpg,Cheese,0.567861,0.000161,,,0.699543,0.077502,,\n", + "TEST,gs://migration-ucaip-trainingaip-20210301091741/openimage/1064/3167707458_7b2eebed9e_o.jpg,Cheese,0.916052,0.085569,,,1.000000,0.348036,,\n", + "TEST,gs://migration-ucaip-trainingaip-20210301091741/openimage/1064/3167707458_7b2eebed9e_o.jpg,Salad,0.000000,0.000000,,,1.000000,1.000000,,\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_create_dataset:migration" + }, + "source": [ + "## Create a dataset" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_datasets_create:migration,old" + }, + "source": [ + "### [projects.locations.datasets.create](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.datasets/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ozaMOwdNJjkT" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "KxWzIF7_yaa5" + }, + "outputs": [], + "source": [ + "dataset = {\n", + " \"display_name\": \"salads_20210301091741\",\n", + " \"image_object_detection_dataset_metadata\": {},\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " automl.CreateDatasetRequest(parent=PARENT, dataset=dataset).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"dataset\": {\n", + " \"displayName\": \"salads_20210301091741\",\n", + " \"imageObjectDetectionDatasetMetadata\": {}\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "vHkuSZxkJjkT" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "3KJwP6EMyaa6" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].create_dataset(parent=PARENT, dataset=dataset)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "FLqkZMD2JjkT" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "7pD6Cnaayaa6" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/datasets/IOD6853960213125398528\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "3098fbf69816" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_datasets_importdata:migration,old" + }, + "source": [ + "### [projects.locations.datasets.importData](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.datasets/importData)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "UI4gtgi8yaa7" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "C2WaWWRdyaa7" + }, + "outputs": [], + "source": [ + "input_config = {\"gcs_source\": {\"input_uris\": [IMPORT_FILE]}}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " automl.ImportDataRequest(name=dataset_id, input_config=input_config).__dict__[\n", + " \"_pb\"\n", + " ]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/datasets/IOD6853960213125398528\",\n", + " \"inputConfig\": {\n", + " \"gcsSource\": {\n", + " \"inputUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210301091741/openimage/salads_ml_use.csv\"\n", + " ]\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Maf1LoqCyaa7" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "QdxyXocIyaa8" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].import_data(name=dataset_id, input_config=input_config)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "yYvi8E5gyaa8" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "4csMUj2Nyaa8" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_create_and_deploy_model:migration" + }, + "source": [ + "## Train a model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_models_create:migration,old" + }, + "source": [ + "### [projects.locations.models.create](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "IL7EQThtyaa9" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "mhWvSo8Lyaa9" + }, + "outputs": [], + "source": [ + "model = {\n", + " \"display_name\": \"salads_\" + TIMESTAMP,\n", + " \"dataset_id\": dataset_short_id,\n", + " \"image_object_detection_model_metadata\": {\"train_budget_milli_node_hours\": 20000},\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(automl.CreateModelRequest(parent=PARENT, model=model).__dict__[\"_pb\"])\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"model\": {\n", + " \"displayName\": \"salads_20210301091741\",\n", + " \"datasetId\": \"IOD6853960213125398528\",\n", + " \"imageObjectDetectionModelMetadata\": {\n", + " \"trainBudgetMilliNodeHours\": \"20000\"\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "9SVvg6Fuyaa9" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "G0CsKetKyaa9" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].create_model(parent=PARENT, model=model)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "A5cpYaRHyaa-" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "hD6pFBA2yaa-" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/IOD3797407498105782272\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "c16f1dfb6421" + }, + "outputs": [], + "source": [ + "# The full unique ID for the training pipeline\n", + "model_id = result.name\n", + "# The short numeric ID for the training pipeline\n", + "model_short_id = model_id.split(\"/\")[-1]\n", + "\n", + "print(model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_evaluate_the_model:migration" + }, + "source": [ + "## Evaluate the model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "yfufMwAEJjkX" + }, + "source": [ + "### [projects.locations.models.modelEvaluations.list](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models.modelEvaluations/list)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "4UuWIZrQyaa_" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "8TCpbLglyaa_" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].list_model_evaluations(parent=model_id, filter=\"\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "-LGUxB-Syaa_" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "AxEY7WFHj2YC" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "model_evaluations = [json.loads(MessageToJson(me.__dict__[\"_pb\"])) for me in request]\n", + "# The evaluation slice\n", + "evaluation_slice = request.model_evaluation[0].name\n", + "\n", + "print(json.dumps(model_evaluations, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "[\n", + " {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/IOD3797407498105782272/modelEvaluations/794074077030707821\",\n", + " \"annotationSpecId\": \"6622317852164620288\",\n", + " \"createTime\": \"2021-03-01T10:51:08.382156Z\",\n", + " \"evaluatedExampleCount\": 18,\n", + " \"imageObjectDetectionEvaluationMetrics\": {\n", + " \"evaluatedBoundingBoxCount\": 96,\n", + " \"boundingBoxMetricsEntries\": [\n", + " {\n", + " \"iouThreshold\": 0.1,\n", + " \"meanAveragePrecision\": 0.48460323,\n", + " \"confidenceMetricsEntries\": [\n", + " {\n", + " \"confidenceThreshold\": 3.6381647e-05,\n", + " \"recall\": 0.65625,\n", + " \"precision\": 0.26359832,\n", + " \"f1Score\": 0.3761194\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.00059657835,\n", + " \"recall\": 0.6458333,\n", + " \"precision\": 0.2792793,\n", + " \"f1Score\": 0.38993713\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.0012138822,\n", + " \"recall\": 0.6354167,\n", + " \"precision\": 0.2961165,\n", + " \"f1Score\": 0.4039735\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.003875596,\n", + " \"recall\": 0.625,\n", + " \"precision\": 0.32967034,\n", + " \"f1Score\": 0.4316547\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.0065186527,\n", + " \"recall\": 0.6145833,\n", + " \"precision\": 0.3597561,\n", + " \"f1Score\": 0.45384616\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.008566886,\n", + " \"recall\": 0.6041667,\n", + " \"precision\": 0.36708862,\n", + " \"f1Score\": 0.45669293\n", + " },\n", + " \n", + " # REMOVED FOR BREVITY\n", + " \n", + " {\n", + " \"confidenceThreshold\": 0.024422897,\n", + " \"recall\": 0.1875,\n", + " \"precision\": 0.06818182,\n", + " \"f1Score\": 0.10000001\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.03706667,\n", + " \"recall\": 0.125,\n", + " \"precision\": 0.06896552,\n", + " \"f1Score\": 0.08888888\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.17009574,\n", + " \"recall\": 0.0625,\n", + " \"precision\": 0.16666667,\n", + " \"f1Score\": 0.09090909\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9619299,\n", + " \"recall\": 0.0625,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.11764706\n", + " }\n", + " ]\n", + " },\n", + " {\n", + " \"iouThreshold\": 0.85,\n", + " \"confidenceMetricsEntries\": [\n", + " {\n", + " \"confidenceThreshold\": 4.645326e-05\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9619299\n", + " }\n", + " ]\n", + " }\n", + " ],\n", + " \"boundingBoxMeanAveragePrecision\": 0.08613319\n", + " },\n", + " \"displayName\": \"Baked Goods\"\n", + " }\n", + "]\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "i6T0bzuNJjkY" + }, + "source": [ + "### [projects.locations.models.modelEvaluations.get](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models.modelEvaluations/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aWycptl9yabA" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dVMiln2uyabA" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].get_model_evaluation(name=evaluation_slice)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "MeB0FekxyabA" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "w3-uF8bOyabB" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/IOD3797407498105782272/modelEvaluations/794074077030707821\",\n", + " \"annotationSpecId\": \"6622317852164620288\",\n", + " \"createTime\": \"2021-03-01T10:51:08.382156Z\",\n", + " \"evaluatedExampleCount\": 18,\n", + " \"imageObjectDetectionEvaluationMetrics\": {\n", + " \"evaluatedBoundingBoxCount\": 96,\n", + " \"boundingBoxMetricsEntries\": [\n", + " {\n", + " \"iouThreshold\": 0.1,\n", + " \"meanAveragePrecision\": 0.48460323,\n", + " \"confidenceMetricsEntries\": [\n", + " {\n", + " \"confidenceThreshold\": 3.6381647e-05,\n", + " \"recall\": 0.65625,\n", + " \"precision\": 0.26359832,\n", + " \"f1Score\": 0.3761194\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.00059657835,\n", + " \"recall\": 0.6458333,\n", + " \"precision\": 0.2792793,\n", + " \"f1Score\": 0.38993713\n", + " },\n", + " \n", + " # REMOVED FOR BREVITY\n", + " \n", + " {\n", + " \"confidenceThreshold\": 0.9941245,\n", + " \"recall\": 0.03125,\n", + " \"precision\": 0.6,\n", + " \"f1Score\": 0.05940594\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99589527,\n", + " \"recall\": 0.020833334,\n", + " \"precision\": 0.6666667,\n", + " \"f1Score\": 0.040404044\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9964624,\n", + " \"recall\": 0.010416667,\n", + " \"precision\": 0.5,\n", + " \"f1Score\": 0.020408163\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.9993537,\n", + " \"recall\": 0.010416667,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.020618558\n", + " }\n", + " ]\n", + " }\n", + " ],\n", + " \"boundingBoxMeanAveragePrecision\": 0.30185306\n", + " },\n", + " \"displayName\": \"Tomato\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "A4zJyXMdyabB" + }, + "source": [ + "## Make batch predictions" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "IEFR6ypNyabB" + }, + "source": [ + "### Make a batch prediction file" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "1Ji3-owXyabB" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "import tensorflow as tf\n", + "\n", + "test_items = ! gsutil cat $IMPORT_FILE | head -n 10\n", + "\n", + "gcs_input_uri = \"gs://\" + BUCKET_NAME + \"/test.csv\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " for item in test_items:\n", + " f.write(item.split(\",\")[1] + \"\\n\")\n", + "\n", + "! gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210301091741/openimage/103/279324025_3e74a32a84_o.jpg\n", + "gs://migration-ucaip-trainingaip-20210301091741/openimage/103/279324025_3e74a32a84_o.jpg\n", + "gs://migration-ucaip-trainingaip-20210301091741/openimage/1064/3167707458_7b2eebed9e_o.jpg\n", + "gs://migration-ucaip-trainingaip-20210301091741/openimage/1064/3167707458_7b2eebed9e_o.jpg\n", + "gs://migration-ucaip-trainingaip-20210301091741/openimage/1064/3167707458_7b2eebed9e_o.jpg\n", + "gs://migration-ucaip-trainingaip-20210301091741/openimage/1064/3167707458_7b2eebed9e_o.jpg\n", + "gs://migration-ucaip-trainingaip-20210301091741/openimage/1064/3167707458_7b2eebed9e_o.jpg\n", + "gs://migration-ucaip-trainingaip-20210301091741/openimage/1064/3167707458_7b2eebed9e_o.jpg\n", + "gs://migration-ucaip-trainingaip-20210301091741/openimage/1064/3167707458_7b2eebed9e_o.jpg\n", + "gs://migration-ucaip-trainingaip-20210301091741/openimage/1064/3167707458_7b2eebed9e_o.jpg\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_models_batchpredict:migration,old" + }, + "source": [ + "### [projects.locations.models.batchPredict](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models/batchPredict)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "skd_JJzRyabB" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "2srxmGqkyabC" + }, + "outputs": [], + "source": [ + "input_config = {\"gcs_source\": {\"input_uris\": [gcs_input_uri]}}\n", + "\n", + "output_config = {\n", + " \"gcs_destination\": {\"output_uri_prefix\": \"gs://\" + f\"{BUCKET_NAME}/batch_output/\"}\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " automl.BatchPredictRequest(\n", + " name=model_id, input_config=input_config, output_config=output_config\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0bc2ac22cc96" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/IOD3797407498105782272\",\n", + " \"inputConfig\": {\n", + " \"gcsSource\": {\n", + " \"inputUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210301091741/test.csv\"\n", + " ]\n", + " }\n", + " },\n", + " \"outputConfig\": {\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210301091741/batch_output/\"\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "46cbhLkryabC" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "2NDzN1n8yabC" + }, + "outputs": [], + "source": [ + "request = clients[\"prediction\"].batch_predict(\n", + " name=model_id, input_config=input_config, output_config=output_config\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "zP-ACCH9yabC" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Xaf2gIubyabC" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "DYzaoP0CEV7j" + }, + "source": [ + "*Example output*:\r\n", + "```\r\n", + "{}\r\n", + "```" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "fnYRCde3xXie" + }, + "outputs": [], + "source": [ + "destination_uri = output_config[\"gcs_destination\"][\"output_uri_prefix\"][:-1]\n", + "\n", + "! gsutil ls $destination_uri/*\n", + "! gsutil cat $destination_uri/prediction*/*.jsonl" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "fe1abc8b829e" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210301091741/batch_output/prediction-salads_20210301091741-2021-03-01T10:52:53.972802Z/image_object_detection_0.jsonl\n", + "gs://migration-ucaip-trainingaip-20210301091741/batch_output/prediction-salads_20210301091741-2021-03-01T10:52:53.972802Z/image_object_detection_1.jsonl\n", + "gs://migration-ucaip-trainingaip-20210301091741/batch_output/prediction-salads_20210301091741-2021-03-01T10:52:53.972802Z/image_object_detection_2.jsonl\n", + "{\"ID\":\"gs://migration-ucaip-trainingaip-20210301091741/openimage/103/279324025_3e74a32a84_o.jpg\",\"annotations\":[{\"annotation_spec_id\":\"3163553338344079360\",\"display_name\":\"Baked Goods\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.019032001,\"y\":0.047988813},{\"x\":0.5898183,\"y\":0.78811049}],\"vertices\":[]},\"score\":0.96192992}},{\"annotation_spec_id\":\"6622317852164620288\",\"display_name\":\"Tomato\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.55052197,\"y\":0.58303279},{\"x\":0.63364756,\"y\":0.74831831}],\"vertices\":[]},\"score\":0.870511}},{\"annotation_spec_id\":\"3163553338344079360\",\"display_name\":\"Baked Goods\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.056096584},{\"x\":0.91519189,\"y\":0.98487538}],\"vertices\":[]},\"score\":0.79013598}},{\"annotation_spec_id\":\"4316474842950926336\",\"display_name\":\"Salad\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.2624923,\"y\":0.063098952},{\"x\":0.99545622,\"y\":1}],\"vertices\":[]},\"score\":0.60728121}}]}\n", + "{\"ID\":\"gs://migration-ucaip-trainingaip-20210301091741/openimage/103/279324025_3e74a32a84_o.jpg\",\"annotations\":[{\"annotation_spec_id\":\"3163553338344079360\",\"display_name\":\"Baked Goods\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.019032001,\"y\":0.047988813},{\"x\":0.5898183,\"y\":0.78811049}],\"vertices\":[]},\"score\":0.96192992}},{\"annotation_spec_id\":\"6622317852164620288\",\"display_name\":\"Tomato\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.55052197,\"y\":0.58303279},{\"x\":0.63364756,\"y\":0.74831831}],\"vertices\":[]},\"score\":0.870511}},{\"annotation_spec_id\":\"3163553338344079360\",\"display_name\":\"Baked Goods\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.056096584},{\"x\":0.91519189,\"y\":0.98487538}],\"vertices\":[]},\"score\":0.79013598}},{\"annotation_spec_id\":\"4316474842950926336\",\"display_name\":\"Salad\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.2624923,\"y\":0.063098952},{\"x\":0.99545622,\"y\":1}],\"vertices\":[]},\"score\":0.60728121}}]}\n", + "{\"ID\":\"gs://migration-ucaip-trainingaip-20210301091741/openimage/1064/3167707458_7b2eebed9e_o.jpg\",\"annotations\":[{\"annotation_spec_id\":\"2010631833737232384\",\"display_name\":\"Cheese\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.5891307,\"y\":0.055182356},{\"x\":0.88701528,\"y\":0.2786459}],\"vertices\":[]},\"score\":0.99448556}},{\"annotation_spec_id\":\"4316474842950926336\",\"display_name\":\"Salad\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{},{\"x\":0.99119592,\"y\":1}],\"vertices\":[]},\"score\":0.78258878}},{\"annotation_spec_id\":\"2010631833737232384\",\"display_name\":\"Cheese\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.09908694,\"y\":0.038005471},{\"x\":0.53745395,\"y\":0.41128135}],\"vertices\":[]},\"score\":0.61209571}},{\"annotation_spec_id\":\"6622317852164620288\",\"display_name\":\"Tomato\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.56387162,\"y\":0.40369061},{\"x\":0.99384451,\"y\":0.9047623299999999}],\"vertices\":[]},\"score\":0.57341981}},{\"annotation_spec_id\":\"2010631833737232384\",\"display_name\":\"Cheese\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.58524036,\"y\":0.41812626},{\"x\":0.99431562,\"y\":0.8581754}],\"vertices\":[]},\"score\":0.50971383}}]}\n", + "{\"ID\":\"gs://migration-ucaip-trainingaip-20210301091741/openimage/1064/3167707458_7b2eebed9e_o.jpg\",\"annotations\":[{\"annotation_spec_id\":\"2010631833737232384\",\"display_name\":\"Cheese\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.5891307,\"y\":0.055182356},{\"x\":0.88701528,\"y\":0.2786459}],\"vertices\":[]},\"score\":0.99448556}},{\"annotation_spec_id\":\"4316474842950926336\",\"display_name\":\"Salad\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{},{\"x\":0.99119592,\"y\":1}],\"vertices\":[]},\"score\":0.78258878}},{\"annotation_spec_id\":\"2010631833737232384\",\"display_name\":\"Cheese\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.09908694,\"y\":0.038005471},{\"x\":0.53745395,\"y\":0.41128135}],\"vertices\":[]},\"score\":0.61209571}},{\"annotation_spec_id\":\"6622317852164620288\",\"display_name\":\"Tomato\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.56387162,\"y\":0.40369061},{\"x\":0.99384451,\"y\":0.9047623299999999}],\"vertices\":[]},\"score\":0.57341981}},{\"annotation_spec_id\":\"2010631833737232384\",\"display_name\":\"Cheese\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.58524036,\"y\":0.41812626},{\"x\":0.99431562,\"y\":0.8581754}],\"vertices\":[]},\"score\":0.50971383}}]}\n", + "{\"ID\":\"gs://migration-ucaip-trainingaip-20210301091741/openimage/1064/3167707458_7b2eebed9e_o.jpg\",\"annotations\":[{\"annotation_spec_id\":\"2010631833737232384\",\"display_name\":\"Cheese\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.5891307,\"y\":0.055182356},{\"x\":0.88701528,\"y\":0.2786459}],\"vertices\":[]},\"score\":0.99448556}},{\"annotation_spec_id\":\"4316474842950926336\",\"display_name\":\"Salad\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{},{\"x\":0.99119592,\"y\":1}],\"vertices\":[]},\"score\":0.78258878}},{\"annotation_spec_id\":\"2010631833737232384\",\"display_name\":\"Cheese\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.09908694,\"y\":0.038005471},{\"x\":0.53745395,\"y\":0.41128135}],\"vertices\":[]},\"score\":0.61209571}},{\"annotation_spec_id\":\"6622317852164620288\",\"display_name\":\"Tomato\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.56387162,\"y\":0.40369061},{\"x\":0.99384451,\"y\":0.9047623299999999}],\"vertices\":[]},\"score\":0.57341981}},{\"annotation_spec_id\":\"2010631833737232384\",\"display_name\":\"Cheese\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.58524036,\"y\":0.41812626},{\"x\":0.99431562,\"y\":0.8581754}],\"vertices\":[]},\"score\":0.50971383}}]}\n", + "{\"ID\":\"gs://migration-ucaip-trainingaip-20210301091741/openimage/1064/3167707458_7b2eebed9e_o.jpg\",\"annotations\":[{\"annotation_spec_id\":\"2010631833737232384\",\"display_name\":\"Cheese\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.5891307,\"y\":0.055182356},{\"x\":0.88701528,\"y\":0.2786459}],\"vertices\":[]},\"score\":0.99448556}},{\"annotation_spec_id\":\"4316474842950926336\",\"display_name\":\"Salad\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{},{\"x\":0.99119592,\"y\":1}],\"vertices\":[]},\"score\":0.78258878}},{\"annotation_spec_id\":\"2010631833737232384\",\"display_name\":\"Cheese\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.09908694,\"y\":0.038005471},{\"x\":0.53745395,\"y\":0.41128135}],\"vertices\":[]},\"score\":0.61209571}},{\"annotation_spec_id\":\"6622317852164620288\",\"display_name\":\"Tomato\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.56387162,\"y\":0.40369061},{\"x\":0.99384451,\"y\":0.9047623299999999}],\"vertices\":[]},\"score\":0.57341981}},{\"annotation_spec_id\":\"2010631833737232384\",\"display_name\":\"Cheese\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.58524036,\"y\":0.41812626},{\"x\":0.99431562,\"y\":0.8581754}],\"vertices\":[]},\"score\":0.50971383}}]}\n", + "{\"ID\":\"gs://migration-ucaip-trainingaip-20210301091741/openimage/1064/3167707458_7b2eebed9e_o.jpg\",\"annotations\":[{\"annotation_spec_id\":\"2010631833737232384\",\"display_name\":\"Cheese\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.5891307,\"y\":0.055182356},{\"x\":0.88701528,\"y\":0.2786459}],\"vertices\":[]},\"score\":0.99448556}},{\"annotation_spec_id\":\"4316474842950926336\",\"display_name\":\"Salad\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{},{\"x\":0.99119592,\"y\":1}],\"vertices\":[]},\"score\":0.78258878}},{\"annotation_spec_id\":\"2010631833737232384\",\"display_name\":\"Cheese\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.09908694,\"y\":0.038005471},{\"x\":0.53745395,\"y\":0.41128135}],\"vertices\":[]},\"score\":0.61209571}},{\"annotation_spec_id\":\"6622317852164620288\",\"display_name\":\"Tomato\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.56387162,\"y\":0.40369061},{\"x\":0.99384451,\"y\":0.9047623299999999}],\"vertices\":[]},\"score\":0.57341981}},{\"annotation_spec_id\":\"2010631833737232384\",\"display_name\":\"Cheese\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.58524036,\"y\":0.41812626},{\"x\":0.99431562,\"y\":0.8581754}],\"vertices\":[]},\"score\":0.50971383}}]}\n", + "{\"ID\":\"gs://migration-ucaip-trainingaip-20210301091741/openimage/1064/3167707458_7b2eebed9e_o.jpg\",\"annotations\":[{\"annotation_spec_id\":\"2010631833737232384\",\"display_name\":\"Cheese\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.5891307,\"y\":0.055182356},{\"x\":0.88701528,\"y\":0.2786459}],\"vertices\":[]},\"score\":0.99448556}},{\"annotation_spec_id\":\"4316474842950926336\",\"display_name\":\"Salad\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{},{\"x\":0.99119592,\"y\":1}],\"vertices\":[]},\"score\":0.78258878}},{\"annotation_spec_id\":\"2010631833737232384\",\"display_name\":\"Cheese\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.09908694,\"y\":0.038005471},{\"x\":0.53745395,\"y\":0.41128135}],\"vertices\":[]},\"score\":0.61209571}},{\"annotation_spec_id\":\"6622317852164620288\",\"display_name\":\"Tomato\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.56387162,\"y\":0.40369061},{\"x\":0.99384451,\"y\":0.9047623299999999}],\"vertices\":[]},\"score\":0.57341981}},{\"annotation_spec_id\":\"2010631833737232384\",\"display_name\":\"Cheese\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.58524036,\"y\":0.41812626},{\"x\":0.99431562,\"y\":0.8581754}],\"vertices\":[]},\"score\":0.50971383}}]}\n", + "{\"ID\":\"gs://migration-ucaip-trainingaip-20210301091741/openimage/1064/3167707458_7b2eebed9e_o.jpg\",\"annotations\":[{\"annotation_spec_id\":\"2010631833737232384\",\"display_name\":\"Cheese\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.5891307,\"y\":0.055182356},{\"x\":0.88701528,\"y\":0.2786459}],\"vertices\":[]},\"score\":0.99448556}},{\"annotation_spec_id\":\"4316474842950926336\",\"display_name\":\"Salad\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{},{\"x\":0.99119592,\"y\":1}],\"vertices\":[]},\"score\":0.78258878}},{\"annotation_spec_id\":\"2010631833737232384\",\"display_name\":\"Cheese\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.09908694,\"y\":0.038005471},{\"x\":0.53745395,\"y\":0.41128135}],\"vertices\":[]},\"score\":0.61209571}},{\"annotation_spec_id\":\"6622317852164620288\",\"display_name\":\"Tomato\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.56387162,\"y\":0.40369061},{\"x\":0.99384451,\"y\":0.9047623299999999}],\"vertices\":[]},\"score\":0.57341981}},{\"annotation_spec_id\":\"2010631833737232384\",\"display_name\":\"Cheese\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.58524036,\"y\":0.41812626},{\"x\":0.99431562,\"y\":0.8581754}],\"vertices\":[]},\"score\":0.50971383}}]}\n", + "{\"ID\":\"gs://migration-ucaip-trainingaip-20210301091741/openimage/1064/3167707458_7b2eebed9e_o.jpg\",\"annotations\":[{\"annotation_spec_id\":\"2010631833737232384\",\"display_name\":\"Cheese\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.5891307,\"y\":0.055182356},{\"x\":0.88701528,\"y\":0.2786459}],\"vertices\":[]},\"score\":0.99448556}},{\"annotation_spec_id\":\"4316474842950926336\",\"display_name\":\"Salad\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{},{\"x\":0.99119592,\"y\":1}],\"vertices\":[]},\"score\":0.78258878}},{\"annotation_spec_id\":\"2010631833737232384\",\"display_name\":\"Cheese\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.09908694,\"y\":0.038005471},{\"x\":0.53745395,\"y\":0.41128135}],\"vertices\":[]},\"score\":0.61209571}},{\"annotation_spec_id\":\"6622317852164620288\",\"display_name\":\"Tomato\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.56387162,\"y\":0.40369061},{\"x\":0.99384451,\"y\":0.9047623299999999}],\"vertices\":[]},\"score\":0.57341981}},{\"annotation_spec_id\":\"2010631833737232384\",\"display_name\":\"Cheese\",\"image_object_detection\":{\"bounding_box\":{\"normalized_vertices\":[{\"x\":0.58524036,\"y\":0.41812626},{\"x\":0.99431562,\"y\":0.8581754}],\"vertices\":[]},\"score\":0.50971383}}]}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "DG45Ug0jyabD" + }, + "source": [ + "## Make online predictions" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "v3Hc1PEyyabD" + }, + "source": [ + "### Prepare file for online prediction" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "FGkjkIRsyabD" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "test_item = ! gsutil cat $IMPORT_FILE | head -n 1\n", + "test_file = test_item[0].split(\",\")[1]\n", + "\n", + "with tf.io.gfile.GFile(test_file, \"rb\") as f:\n", + " content = f.read()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_models_deploy:migration,old" + }, + "source": [ + "### [projects.locations.models.deploy](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models/deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Rlct02LHyabE" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "0b4367e8c21e" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].deploy_model(name=model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "5z_mpZ82yabE" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "IJVSEo-nyabE" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "DYzaoP0CEV7j" + }, + "source": [ + "*Example output*:\r\n", + "```\r\n", + "{}\r\n", + "```" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_models_predict:migration,old" + }, + "source": [ + "### [projects.locations.models.predict](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models/predict)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Rzs-u8IRyabF" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "uLUIi93fyabF" + }, + "outputs": [], + "source": [ + "image = automl.Image(image_bytes=content)\n", + "payload = automl.ExamplePayload(image=image)\n", + "params = {\"score_threshold\": \"0.8\"}\n", + "\n", + "prediction_request = automl.PredictRequest(\n", + " name=model_id, payload=payload, params=params\n", + ")\n", + "\n", + "print(MessageToJson(prediction_request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "3409334bbfc7" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/IOD3797407498105782272\",\n", + " \"payload\": {\n", + " \"image\": {\n", + " \"imageBytes\": \"/9j/4RtSRXhpZgAASUkqAAgAAAAIAA8BAgAGAAAAbgAAABABAgAEAAAATjkxABIBAwABAAAAAQAAABoBBQABAAAAdAAAABsBBQABAAAAfAAAACgBAwABAAAAAgAAABMCAwABAAAAAQAAAGmHBAABAAAAhAAAAOoBAABOb2tpYQAsAQAAAQAAACwBAAABAAAAFgCaggUAAQAAAJIBAACdggUAAQAAAJoBAAAniAMAAQAAAH0AAAAAkAcABAAAADAyMjADkAIAFAAAAKIBAAAEkAIAFAAAALYBAAABkQcABAAAAAECAwABkgoAAQAAAMoBAAACkgUAAQAAANIBAAAIkgMAAQAAAAAAAAAJkgMAAQAAACAAAAAKkgUAAQAAANoBAAAAoAcABAAAADAxMDABoAMAAQAAAAEAAAACoAQAAQAAAEAGAAADoAQAAQAAALAEAAABpAMAAQAAAAAAAAACpAMAAQAAAAAAAAADpAMAAQAAAAAAAAAEpAUAAQAAAOIBAAAGpAMAAQAAAAAAAAAHpAMAAQAAAAEAAAAAAAAATecAAEBCDwAgAAAACgAAADIwMDY6MTA6MjUgMTI6MzM6MjAAMjAwNjoxMDoyNSAxMjozMzoyMADtDwAA6AMAAFABAABkAAAALQAAAAoAAABkAAAAZAAAAAYAAwEDAAEAAAAGAAAAGgEFAAEAAAA4AgAAGwEFAAEAAABAAgAAKAEDAAEAAAACAAAAAQIEAAEAAABIAgAAAgIEAAEAAAACGQAAAAAAAEgAAAABAAAASAAAAAEAAAD/2P/bAIQABQMEBAQDBQQEBAUFBQYHDAgHBwcHDwsLCQwRDxISEQ8RERMWHBcTFBoVEREYIRgaHR0fHx8TFyIkIh4kHB4fHgEFBQUHBgcOCAgOHhQRFB4eHh4eHh4eHh4eHh4eHh4eHh4eHh4eHh4eHh4eHh4eHh4eHh4eHh4eHh4eHh4eHh4e/8AAEQgAeACgAwEhAAIRAQMRAf/EAaIAAAEFAQEBAQEBAAAAAAAAAAABAgMEBQYHCAkKCxAAAgEDAwIEAwUFBAQAAAF9AQIDAAQRBRIhMUEGE1FhByJxFDKBkaEII0KxwRVS0fAkM2JyggkKFhcYGRolJicoKSo0NTY3ODk6Q0RFRkdISUpTVFVWV1hZWmNkZWZnaGlqc3R1dnd4eXqDhIWGh4iJipKTlJWWl5iZmqKjpKWmp6ipqrKztLW2t7i5usLDxMXGx8jJytLT1NXW19jZ2uHi4+Tl5ufo6erx8vP09fb3+Pn6AQADAQEBAQEBAQEBAAAAAAAAAQIDBAUGBwgJCgsRAAIBAgQEAwQHBQQEAAECdwABAgMRBAUhMQYSQVEHYXETIjKBCBRCkaGxwQkjM1LwFWJy0QoWJDThJfEXGBkaJicoKSo1Njc4OTpDREVGR0hJSlNUVVZXWFlaY2RlZmdoaWpzdHV2d3h5eoKDhIWGh4iJipKTlJWWl5iZmqKjpKWmp6ipqrKztLW2t7i5usLDxMXGx8jJytLT1NXW19jZ2uLj5OXm5+jp6vLz9PX29/j5+v/aAAwDAQACEQMRAD8A85mvUAy8iLjuTUH9qxY+WRjx26V50YM2samgaRqetkzeSLWyXkzSHlh/sirN/qmk6BG8UStI4HLAdT7n/CtKc4RlYp03a5xsup3msajuWKSSSQ7Y448k47YAr0Dwt8IfGeulJp7dNMt2/juT82PZRz/KtFGVR6GbaitT0/QfgPoNsitquo3V646qmI0/xrsNM+GfgfTwPJ8PWcjD+KZfMJ/OuuGHhHfUxdRvY37bRtKtECW2m2cKjoEhVf5CpjbwgYEcYHsorayWxGrK81pbSKQ8ELj0KA1jX/hjw9dZ+0aLYOfXyFB/PFQ4xlugTaOZv/hh4RmUiGzks+48mQgD8DXI6z8I2jUvpGqq4ByIp12k+24f4VyVMKnrHQ1jVfU868ReG/Feil2vNNeKBRxJH+9U/iOBXNm48x8yB2bGCd3X/PpXJNKl7pqve1LunpaTnO3aR/CR2q/HbxbvkwAe2M1i5SiPlTJTBAikDjPfpUUu6HBUscchY+c0o2bvsTY49mjQ5YAkdycmo31JYWXywsjA9+VH+NdMFzas2hC5ft/G2p26GN7l5ABhIhwAfeu2+Hvwt8T+PrhNX16V9O0w8qWXDuP9hew9zThh1Kd1uVUq2jbofRfgrwP4W8HWoXTLCJZcfPcSjdI34np+FL4m+Ifhnw6xXUb3BGBtjXcfyFek5QoQu9EccYSrSstzi/FPxltTCsfh2GWR2OPNkTCj8KwLD4t+LFuQk62jqemU7d+leLic2lGf7rZHpUsujy/vNzv/AAp8R7XUsQakiWsx6SBvkPp9K61r1WUMrgqRkEHrXoYLGxxULrRrdHFiMO6MrdCnc6nHEpLOABXM6n42t45fItVNw44O08LW9WtGlG7MoU3JmVJr+pXcmzz0hB/hQZNWbfT726UNJLdy59yBXjVcXOponZHTGlGCLwt1sHVZGlBxghuRj3rmvEvgrwprwcwxpZ3r/wDLSEBAT6kdKinVt7s9ROPVHlvjHwJrPhiZyQLy1Xpcw8jHuO1c5BfSwDliy+h6Vvy6WZKZfXUbeaLOVU9+2Kk1CcpaW4iDATRNuYAdQ2R/KsZQ5WrmlPqcT4m077JPta/g8sDOFbLn2x61hh3kdYoEbLfKuB8x9q7l7qsUnpZH0J8FvhHaafbweJfF0KvNxJb2b9E9Cw7n2r1LxF4907S7Mi2HmsnAjj7V1KccPT5pGHI6srI881z4m6zqNuUtEECtx6nGK4e5hN+pe7mdjI2WBOWNfOY3Hzru3RHs4bDRoxutyytphdtunlqnQYzmnBp/tqhY1YAZCkcV5anc6Gk/iLtwYQDGzbGYZGOi+1auh+LdY0e2WzE4khY/uxJltvsPQVtha88NPngZVacasFGRJrXjK7vrNUI8hmbazA/KB7U6wkW3s1aIct0P9a9OrjPrCUkcLoey0O18FWHnYkMIcnks3eu+SWG2gKeW6cc8dKzjokzCerOYvphdXJbaAgPynqSP6VVv9PKWLTk7WY8fSlZtMFoQ6bfWllD9kMYlVh+8Lc7vauP8XfDFNZil1DwoiwSn5pLZjhG/3T2P6VdCul7jFKFvePCdYt7+wvZbO9tpbSSJtrxuMNXUeFof7Q0y03AlFuGt+uTlgNvX3OK6q8lGncdFXkea3kCxSGWJix7hjk17V+zp4BhuVHjDWYcxRn/Q4nHBI/jP9K9CkvaSuZy9yJ2vxK8bfZP9AtXBlbjAP3RXk8upTmQiUZLHPPUn1rzs0r8z5Ox34GjZXIvt0hOGOUByeMEVOLyXYGiPmKDkZ7V4rgkeuopIv2Ws3DQtCybMdx3qZdSgQqqnLE885rOyWxhKnZ6FiO6hni2v0wR8wptukW3HmbiTk5PIA/lUt2RPK0Qa6nnWIWLCgHDewx14rc8GeZqGhQEtukibYwHfB5rbD/C4mGItypnrOj3CWwjKyGOJU5xwelXxcz3ELMhfB/vHtXoI84yL7VrfTJI3uoneFm2mRBkKff2965/xD4plvWaOzBMSYHy1nUnyxaW44Ru9Snpc4mDKd5n7KOa9P8DJqBgVTGkcbDBG3rWNF3d0VV0Wpn/GL4b2PjDw5c/Z7dI9btl3wSBQDJj+An0NfKGk6hcaBc32jXiSwMWDjK/OkiE4r0YR9pTcWZ0H+8RneB9Ck8R+KbLSkB2yvmQjsg5J/KvqbxPe2fhjwr5FuFjjgiCRIvHQYAr1MP7sZSZNVc0lE+e5r1r3UZLqUFpZXzIT1A7AVdFjI9sJyc8ZA718viqjcubue7QSgkhsluLZPNkwFIqKfUdKWMxhxvA42risoQnPY2cr7GZumv7jbDJ5RU5Xn7wqNmlE22T5QpxkVvotOpUUbVq0qRYSQSHb0NIt1KsuyRhGT/EK5JK7DlRamvooW+VgysvOe9X/AIYa5Fa3l5Aowqy7gD2BFb4ZcrucWIjzRPSJtct7mWNIWWPH944z+FXrfW5V3RO6Km3AOcg11e01utjznBpai6Y1je3KC/vlEQcBlHG70rsri30KwCRWccEpbho1AY/j70oygk5XuzKV27E3hXwXYnWpdZSARArgRAAAk+vrXoWjabEiFcBAOgHaroUlBerbIqSci7LbwINrkM3qByK+Vf2kPAlrc6nd6npybLi3O6QKOWX1+o/l9K7aLSukZxbTucr+zZpsccmo69OBuQCGInt3P9KT4leJX1XWJNPifdFEc8Hoa6cRP2eG06nXQhz1vQ5W0RRMUyAT3C8/jW/YGJF8t5WKlcE4r5mom3oey7KNipe2xunwc+UOEHasK90VSf3Z3EHtV06jpvyHBq1ifSlure6CmFGVRjkfyq7eSRTIyG0VXHANRNx5ro1tdqzIosQIr7lAA/yKydd1GJYmeL7w6VVKHPJEylZ3RzUusSuhBZs9MVoeG7m7s3e4dmVpiNifxfWvQlTjCDOScuZnqXgfRrm9gutQlnka6EeYI8jB+v4VYvT9nuij3DHDbQBXi4m8VdMmTdWLsti7pduhlXyyeeTmu88OrFYwfaZpCuGALFunpWWFhKN5S2R5slbQ9Q8N6gTaxsjq4YZxXU2l07JhSFz6ivfhK6TRg0WYkm372VWHqK8g+KckZ8RyARYGwK4YcN/9bFdWHunqT1Pl/wAIeJbvR/BcGnWoP2i5Z3Zs9OcVTtkZ5zKC+8/fOcnNZ42vf3eiPYwtKyv3NW3iHmKA48wjoavxQs8JdlKlT/D0rx3J3OuSEnklWIiPLMRwTxip9OgiaGLzSHcHd061TWhne2wSkCRvLQOc5Ax09qJFzE+yFM8kkjgVhKy6GkXpucXrM8gl8nYc9wDXN6g3zbeeuK9LD6JDqaLQ7Dwv8PrkyJfXd7E6OuY1jGeD71qavpMdgUcuWMYwOMCvFqZsq+I5YqyRzJPRM9J+E2sLalilrFcLIADv6AemKZ8SPDl/HcNqOjW8YglXzHjiBYxnvXRVXtY2S2Y8NONKq1J6NWOB0rxhZWc8aXEkkpVhuCgg13ba3NrujS29pAIPNH8Kk4AOaqcXRwzckXisu9nTVaMro7H4eazNpqw215KHMY2g8gle3Fe5aDNBeWaTxMrAjtW2U1va0+V7nlYul7KRrbooo2Ytj1rwX4i363uv3EyEbc7Vx6DivbpJJHLHVnyXpd08MFnJ91VTapI6810NjJNKpKARxls5ZefrXn4le82z3sO1yo6OOytzArgsZSMFl6VKAsduSzDbnpnk15yk09S27lS6Ybi3RT274qBbhzApT5FyAOMdKuU1bQIwNG1ugLchhjHJyBn61n3t7J5ZRdxAOcCs01IuMLMxdQlQRtPJEN4FYmmacl+z3U8bNBvChQcH/wDVXRSl7ODkZ4qXJE9Q8NxR2nhaaSO98tbc/uYWOePTJ5+lY13Fc6tfKLJlaMIfNaQ5Az0H1ry/q8adZVF1u36nFQcqlZJnXfDi+hhhazliEXlfL5qKcMa9cuotKuPDLPceYiJH+8kib5h+XevRo8jTQ69OVOaZ8wanpEa+J7qPSZXk08zbY2YAFl9+K9Y8H2n2GGF42VAFAZAudw71z4munaPQ9GriJPDckzfvIdNnuoW0+Rbd3fJJPOfQdgK3/DniO80XULm0inWYDjDtx+lSmsPVVSlt2+R59KCxEJRludPe+NDJ4fminVYrpsqAhzx615Rqt4JZWIOfSvpaE3KmpPqefycrZ4d4W0o6/wCBJ1t1VrvTpywHcoRmk08SRNFEXwFABOc59qyx0bJM9LByWqOis5JWQqvyKD09falupJEcsXwD0APSvFb1O/kVyqsx+XnIPdv8atwRBwwULu6DnpSloh2tqAQmcRZG5Rgrmm3tq2+Ip8oPUY5FZcziyrrQxvEEJSOWKFfMOwsPwp3hvQJxaxGa5WFMbpMngd66VJeyt3ODFu6sd5pGl6Y2lyzQ34nt4lJlbIIz6CsfTtR0lz9jiZBICzn+6xHv3rndF811sGBko3vub1rdgwJa2xjjYnLEYAA+tVviJ4xm0fRU0mG4/wBKlTBKnhV/r9a0w2k2kLGVpOKR5v4bvLldTtZJp1nBIZyrZwM9D716Xq+vrDoB/s+5UXEw2RN2BrDG0kp3S0M69Tnir7panS+BoYW8MzXExMzoAu9mOQ1LBptqLtJ1uGjd5CGG/wCUe9cyiqkLvT/hznw2JlQq3RbP2y50h7s/6tCV9zg4rCcvu+YMPqK+pwrfsY83Yqs4ObcdjyH4I6yumeMhYXDgW9+nlnPQMOn9RXqfi7wQJp/tlgAgxkoo4+or0qtH21Gy3WxlTq+yqKRw08N1C7wyRyqVPzZXFV7VrqS8cTxhkHAx0WvnZRSunuj6Cm1JXRPLCSgJ2x8ZVcjn3qTTJp5bhdmdo6nFYT92OoJXRPeAwX/mCLaH5aT2qe5v7drY7mDP7HtWfLzaojldkcvrmrEy7mKRqp+QDqfrW54c1Kxvb6O2vY0WQxhjHk7W/A9fpWteLhQ93c4MVZWO18TzRR6OllDCLeF49zlP4vwFeTW0oWZjDlY1Yhfp3qcG/bRcuhrlqjzanVeFb+ERyS3ZyqyBTIc8Z6fSu7k8AWHiK5tr3UIxcK0QCbZSN3tx6VnONSlK8Tmxy9nUcXsat/4E8IeFtLE0mmzujvs6F2yQcdOlcD4b8NapPqv2S6iT+zpZDjccNjPBHoaq3No9bnEp3TbPoO88L6Nofgp5Li4t7GwCKXaVsY9ya8V8e6lFo9r9o0bXNI1JHcKIoZ9z4PsK7JZeraOxzxqa3Z6ToEk2ieGtKd7WO6urpAzW7AlV4yf1p2v6d4g8VZvfslvD9mjyIgoQ7ev416dKXL+7S2LTXxM+MWMtvcx3MDbJYmDow9RX1H8NvEkPifwxBdhl85F2zJnkMOor06EtC5I0dU0bT9QjZZ4Rkj7w4Nc7P4JtowTazEZ6huc1z4rAQr67M3w+KnS06HNXXg7Ube63eSZYR3QZpLi1NgqyLEU52kkEAV4OIwlSk7NHr0sTCqkQspuUe3iha4dhyFH+cVjSeGtSjD74xAhz8pbOKvCYWco7aCrYiFLTqcnqdlJaX+ZUJVcABuR1r2PRfBuj39lBrs0befFENuGwo444rSdFVJ9rJnkVpt6s5fVtZEunywyud6RlVOP4gTxXH2ilrYHIDYPFcuAiow5TqwE+Wd2WPC1xNDdXkKqJGkVQqsOAc9fTpXuPw50S/t7oSm8edJFyYw3yqccEVWIw/tKiSeo8zXNV5zo9WtJfJ8qWSSUIOC7ZNYluix3EbtgKrcn2rWpSUFZHlp3Oh1PV4fFFhBoGrIPsAOGB53D3+lc3pXwt8N2HiD7fNYudNVwbdGzvkI7c9BXFgcX9cnK91Z/gh1aMYKx6RdadD5j6jcMkMEablTsmBwK5XTvFctjDdRr+9nmVk8xmPAPQ/hXuwjaV0ZRXMj47vW8vCj5pGHAFb/wv8V3HhLXVkkctZzsBOvp6MPpXoUlyx16mrd2fS9re29/bJd2rq0MqhlKnIwfSlL4rpEAlx3psrJIu11Vh6EZqGrlXK0gjAO1FX6DFYOsRgqxqJqyHF6nm3iu0WXcV+8vNegeHtYkh8FhFjZsqoJUfd4rw6lTkbfqb1Y3icutpuZ7i3gjCyZ5PUE9etczcWcCXi2sNx5gZ/mYchfbPevMpVpRbaWhNOTi7o6zw7o8QZQFI7nivU/Ccl3aKUiZSp7sM4Fc6rzVXmW5pUmp7lzxhrkOl2izzwtNLIcLGg6/4CvMLrVdSvtSDysIYSciKPoo9zV47Fv2bh1MKcOp1+jpO80KruID9fwzXXXHiOaK1t4tiu8K4V2/h+g9a4OH1U9rKT1TImkzndY1u6uYAksrMFyTz1zVDSDBFb3GvamNum2I3MOnmv/DGPcn9M19rQipT9CZe7A+WLO5jk3GRgJT3P9KtNASmegNdU7pgtjtvhh46m8OzLp1+Wk05zwe8R/wr3W0u7e9tkubaVZInGVZTkEVvB3QhWao99NsaGswIqhqEXmRNj0rOb0Gjh76zMt00ezJPFdLonmWuhtpSRRsjkFmI+bjsPavm8TUcJNHTUd4nI/Eae4s7W3tI3EfmAllXrgdKw/CNuZMEpvZ2OPwritanZmk4xp4RT6s9Z8P2Y+zgBcMR1rYi8TtomY4dLtbgL8rO5O4mpptUfetc8+MlUdjnfFHiuPVw7NYrbuFwoLZAP5VjaWqzylYZj5wH8PcdCK87FwnUblFHXF8isj1nw5YRDTIWXO8qQ/1rmtbaNJplBJUMdtellmH9jTuupzxd5MpaRp02ovJLLKLeygG+ed/uxr/j7VxvxD8Ux6xPDpulbodGtP8AUIeDK2OZGHqf0FfUUIclLm7mdSV5W7Hz6k+0gsisR6jrV62vQJMqQqHqjdq7LKSsxJ2LyzW8vBIVz054P411HgvxhqfhiYRgme0Jy8DHp7r6Gs0nBjue2eG/EeleIbYS2E48wDLxNw6/hWi+R7Vo+407kRNRSMehrJ6FozpbVGlLrgMa0bKFI4QWXBA6+teNisPefMipS0scT43to5ZTcSxhnC4GR0rI0O+TTNSje2s/NSOHy2weNx5z+tebVap6yLmp14xpo6uPxDOtoTGgi4wNvJAqG11ee5WO2trSSaLJ3lwNzE965KWIUnotuhjWw0sNNKQ64SKRSDEASPypugRLaaxDIThCcE/Wu+VBW0NWtLnrVjcR2mkglstyVH1rCutOhhgOq6/dDT9P6gkZkm9kXv8Ayr1MFhlKK7I5ebluzg/GPi6XVYFsdOi+waPHJiK3VvmkIHDP6n9BXFasqxXCuowrjd9c/wD1/wCdeoveujJaHj455pSOKtI0BWK8A8elSxXkqcb2AFPS1mF7GppmsXlpdxzQyPFKnKyRttIr1Lwv8W/KK2uvR/aI+nnxrhx9R0NNRtoFz0jSNZ0TW0DaXqcErkZ8osFcfgatzQzR8PGR+FZzg0XGaZVPXOOlJJK+zbuxXJJGtrmHqdobjKsRzWFLo9za2hEC+ZhywCjnmvHx+FlVj7p2YWrGlLmZqaXaP9lRpo2ViOUPar1rHLDMHhwmOMYrmweXeyfM92LFVo4izsSwabdXLiO3gkkJ9BmtuHwdPaRrda7e2uj2x53XMmGP0Xqa9ujhHPV7HHOtGCsRar4+8P6JGLXw3byandLx9svRhFP+xH3+p/KuE1XWdS1e/OoapdS3Er4BZjwo9AOw+ldz5Yrkjscl7u7KUwaedQuNo53Yx+FReJIjBp63WUBhOR7r0IpU9GDP/9n/2wCEAAUDAwQDAwUEBAQFBQUFBwsHBwYGBw4KCggLEA4RERAOEA8SFBoWEhMYEw8QFx8XGBscHR0dERYgIiAcIhocHRwBBQUFBwYHDQcHDRwTEBMcHBwcHBwcHBwcHBwcHBwcHBwcHBwcHBwcHBwcHBwcHBwcHBwcHBwcHBwcHBwcHBwcHP/AABEIBLAGQAMBIQACEQEDEQH/xAGiAAABBQEBAQEBAQAAAAAAAAAAAQIDBAUGBwgJCgsQAAIBAwMCBAMFBQQEAAABfQECAwAEEQUSITFBBhNRYQcicRQygZGhCCNCscEVUtHwJDNicoIJChYXGBkaJSYnKCkqNDU2Nzg5OkNERUZHSElKU1RVVldYWVpjZGVmZ2hpanN0dXZ3eHl6g4SFhoeIiYqSk5SVlpeYmZqio6Slpqeoqaqys7S1tre4ubrCw8TFxsfIycrS09TV1tfY2drh4uPk5ebn6Onq8fLz9PX29/j5+gEAAwEBAQEBAQEBAQAAAAAAAAECAwQFBgcICQoLEQACAQIEBAMEBwUEBAABAncAAQIDEQQFITEGEkFRB2FxEyIygQgUQpGhscEJIzNS8BVictEKFiQ04SXxFxgZGiYnKCkqNTY3ODk6Q0RFRkdISUpTVFVWV1hZWmNkZWZnaGlqc3R1dnd4eXqCg4SFhoeIiYqSk5SVlpeYmZqio6Slpqeoqaqys7S1tre4ubrCw8TFxsfIycrS09TV1tfY2dri4+Tl5ufo6ery8/T19vf4+fr/2gAMAwEAAhEDEQA/AON2AnkUGIGvNkkaeQzySORUTKAckVDd2CdmJnJ5FNOc8CiyTsU9AbI60gYtS8wSQ4E08A+tVN7A9UIxK+9CyGrj3YrE6uuKQOM1d0FtRwcjmjzQx5NZt62ZVtRCykcGmE460N2VkLfcQNuHHamlyOpo6ie4wt3zSbifSqavuUIgyeaa6Z6Ck/Jk9bkeGBzmm4Z85NWrB1uNO7OB0FSxOcYNROKaH0GyOQx5qFpTjJIzVK1h3G/aCvPFI92euadk2GjGtcbuTinQbpjhVNVZbITWhrWOlsxDGty2s1jHvWkYrqZydyyE2jjrSsTjmt9FsFtrCcUKeau4OIpBPSmsCKWltSbW0IbiYqhOa5bWL0MxXd0NY1bWsjWEbGQZFzktT0dSeDXNGPQtyB73yjwaYdWB71oqKe4oyHpqHmKR3prPKRlTVRhGL1E2PWKZ1znmo2tmB+Y1pJxSuRuIgCGnPdsBgVKmVayuV5blnHPWoi2TUO0tEEUh4yOad5rA5qW7PcF2GOzOc5ojznmqbvqCJyMLkUwMV5rPd6lW0HbsjmhVPWpcVF3HFK5Mg+XJqORc80rW1B6MagxSNz1NaIi5EUyc0+PAGGoaRa1GSxBjwahZClRawgV2FIz1fKhu26HK5ParEUMjHIUnNRdx2BRuW00+eVR+5Y5/2aevhi+uDhYW/KiUVJ3RUVZ6ss23gLUZiP3JH1q/D8Lr6dhlSPXirjT7B7qWpr2fwinYgspx9K27T4SFcfIxPfNX7Anmj0NW1+EqDBMZJrTtfhRGo/1AA71pGhFqzREqttEadt8M7eI5MS8j86uJ8OIFTPlpx0FaqhbVEOqyxF8PLdOfLHvxVmLwFbA58sCtHTS2JdR21LC+CrZSMRZPuKmXwfbKwBjAB9aapqxLqMlHhS3Gcxj8qenhm3BJEYz9KqFNJC5nckXw7CB9wflTh4et+TsU59BQoroS5MUaFDn/AFa0f2FCrZ8oEH0Gapxitg577h/YUBPEXT1pDoMI58tTSsth8ztcU6DAcHZkfhTRoMDHDRr+VTyoOdjG8PwE8xg0w+HbcnJjX8qpQQ3K6GnwzasThAR9Khl8K2z/APLJQP51Dpoam0Qt4QtD1hXnvTJPBlrn7gNTKimONQqTeBLdmyUHtkZqtL8Pbdv4FOfbmodCL0L9qUpvhrC2T5QyO+KoXHwxgOf3INRPDrYpVm0Z1x8KoWb/AFWKz5vhSFJIVhisHhUWqlyhdfDGUcrxVKX4eXcYOF57Z4rP6toNyRm3fge/jH+rz9KpSeF72Ecwt+VZTpuK0Kik9iFNMuYWAaNh9RVtrPEeWWlHVWZnJNbEUYaIEheK5vxEWklJ/SuulbkM9LmPDGS9a9lbYbJ4om3bQ0WkjR3bMCpgyuvvXBLaxonqOiQbs84qWQZX1qorkixMrJO0T7QauGMzAE1T1jYmSSBrVVQnOKzLmURvjrTjBoE7ilmlXgVErNGc80n7r0K0S0J9zSdOpq3axOoyeaGkxOV9CxKuYjWLPujYkGk4pNEx3LujuRICe9al9GkiZGM1ro07FTWtzMKEPjFEmR04rGJSatcgcMP4quadKYWDN2pN2lqD1Rp39+s0PBFY8C73ya1k7kRXLoXhhBjNVLiJi2eorHfULix7lGCKRpMsRTmkwW5J5SsvTk0kNqY2zjisoR1ugb7mlaxtjJ4FJeXVvbqdz8+ld1Om2zKe2hzmo34nYhKoJaPI2a6n7qsTFuK1LOPs61PBrCxDpnFc1WLmjop7alg65uA5+tQ3OqhhwK41R1TKvqZk05Y5NIk+O1acq1DSw4zHnFSwtuHJNWoW1IlqBnZG5PFOldJF6c+tDhzInm10KTZVjU0Mp7nNUoqxLnfQtC5LJtUc1Wl35yRijlu9xpj4d7d6sxs8bYbkGritBNoilIL01sj6VUXZ2Jv0JoFUj5hUyWZkbKvg+hqr3YrtEjQvGPnwaSIktWThdXQ+a6JjEzDqKia3YDIOTWd5RY12GqsinJFDXBHB6UnO+jY5JPQrl/m9KVZtrc9KSV9GNRstCwL0j7vSl89pBkmp9mk7jTSLNtKAOWFT/a1B5qEtRp6DDeLuwVzUU6CY5UU5KMdSEralY27A05VdeTU03/MVdNj1diMGpY8YzTcl0LtclWMN1NTR/uue1K/Qgc10oPPFI9xE47ZrVStsS12IRtZuKsQIqyKTzzW1OKepHJZFvUEDQgjHrWQCEbgV1pW2JSLqsT1IFPycV5d3fU2FHI5NQyxkZxWsbMGyA8U4VDWtypMVlGMnrURbDUluJNMcH4yKUE9zT23HewjuWHBqNiyitNlYI9hVcjnvSmQnnpUz1C2oLO3ekaU4p2VyXfmFjkJ6nNS780NlMaGwetKzhuT+VFri1I2PtTA2Kq9x3HbsDNKG45pasNxCQQR3pjZApXtuF9SMtzg00uV5Aod27Im1mNZ9w96jxgetVboVuNKZGeKjZCfrRG97Dv0LVppjzgErW/pujrHyQMgVtCLbE5aGtFAEHSnk4Fb8upm+4KTS7cnmrceqEmKE5xQQq0JdAVxpcY61E8uOTVvQb1RjaxqQgQ1xWoakHlJB6mue15amjVlcom7YHANKlzIT8pppcrJuiwgll65NOa3K8mnIXUdG209KtRzYrKTsUiVLor0ps1wX9qzc77jt1K7OepNBORz1qW+wSfQbigAb+lP0BblhQCuKZKm2oe9mMgGd1SAk9q0sSyTeAuMVG5Oc0ithQBjNKHA+tDTbGlccJsA1E05zmhdgb1sKkm45NOLjuKOTsw0Q4AMM0ySMnkAk1NmDeosdpcOQAhOfar0fhu9uACImAJx0607X2FbTU1bL4fXlwuWjYfUYrXs/hRPMfnXOffmtPZyerC6Rs2XwfGRujJ/CugsPhRDHj9wvt71pHDaD9q/sm5Z/DKFAMxKPwrVtvh9bp1RTj2rpp0UjCVRmnbeCbVF/1Y49KvQ+FbdP+WSj8OtX7NLUh1LlqLw9Ch4iXAqwmkRIcBRn6Vpy6aEp3Jl09Mj5eO9SCxRW4UflTtbYTY/7Mo/hHFL5AxQ0S20Hkj2pTFg0mFrjSmT60bM9qLIAC45Joxiq6ABUde9G32pXsAmOaXHBpWAAAetIQDSAaRzSBcGm0AbfQ4oHXtSSKQm0ZJIHNNKjrQAmwUBQetNEdQK/nSGNaSQDXiUD1zTDDu460eo1tYYbYA8qKjaxTuo/Kh26Be2hE+lwN1QflVaTQbdznYPelp1QKTK03hi2lz+6X34qjN4Nt3ziMD8Kj2cbamnO0zPuPAMLnIVT+FZl18OoyeYlJ+lYuia+07mdc/DpMH93+QrmNd+FMlxwgdfepdO2hSnd2Zy0/wALdQtpMphh7ikk8K3lmozCx/Cs6i9123NElKWhWudJuFXLRt+VVUgmjbaynr3rzmnF6l3saCLtjz3qlcSsh6038REXd6lUyEtuq5FfFY8d6am07FPUeLx3XFVpIg8m5qvm5iWuUm+SMdqhkdXPApVI2QKOhNakAg1qReWydQKqmu4pIqXZcZAPHtWbN85rOcrDjYms22HIOPerL3DgdyKcKiWlypq2oyCbfJls1akiV1zQ49iW9SlMhU0m9sYB4rGUjQe+7y+uahiuGib1rTm11Jk0K95Ix3CpoZnmI9qaEkrGjFbGRM4posVU7jUyi76GTnYW3iTzcMQB71qE2VvFuZl4rejSd9SZt9DndY8TxxkpARj2Oa5251KW4Y5Y813RXLH3RWsrjIpcda0be6VIiaqaVrsWtitNdeac1UcOTlc157k4s02Hw72471KVOafQeu42SJyOBTY4nzyKzTTJ1sWVtXJyVqUROeK1clYe4klixFQ+S6UosTumPFo8gyVqJrZ1bgGm9H5Epq9jR023xgPip7u3j9BUSXVMezuiBYgvRakELP2p8/KrNlKPcgktSrEkU+OFeppKd0KS7D2iHUUkTMjU+d30El0ZbS4DHD1NGseckVfPd3ZXLZ3QsqKVylQpKynkZFVZPQh3vZk6xrKMgVG9jH1xiuaVKSvYeuxWlsT1UVE1rtPSs4t7Bdii3BHFTRWrMMircnZ3GnfcSSB0PTApmCGojO6K6D/LYjNSR8ADPNZyd9hJOxOnzDkZNK4yOlQorcZWkjw3emqj7vaqvoUtC2qlVyWpGcqOaQN3K8jeYR61GyuDndWkJW0ZMn0FjkZDV21nJYZFawm3sybaF28dmtsjpWIk2XIJrujfYz1bZdDkc5pyzk964bXVzdJbkqSnipM+tWlZWJcWQuBk81EWx0qLW1G+xG0pPFNLZNN2QcobjjJ6U5ZO+am11oVuO3980MSRxVx8xS3Iye56mkct2NFlzCWmoB6VgW6dKppJpsb3HxL6mmszKaHFXBbiCQdzzSeYc5NSk7leQnm7qEk5q3G10R1sO3A4yaGai3VlITOeBTXBGPSk7OwhmcmmswLcdKrRMpRutRrZ65xUbvxgGlo9ESloJFHJM4HJrZsNILkMy8Vq47WK0sbtrp6RLwKtrEFFaRWuxi7isp9qTy/U1roA4qOOaQkDvTixNCM/eonck5FUmGxG7bVyao3l6sa9azk/M0iuaxzOsTmbPzZrAa0DNk1EY3dxVXZ2Qf2eByakitlRqG9SIxd2y5FIkY7VHOwc8HrUtJFqPUixtpyk4qWk0Wu4u89KY7sOazkkCeo3ze9Lv3dTS6XQCgnPWnb8daI2Y9xwlI74pjzk0aXFYZ5g/Gpo3B69KcmmtBpD+2RSFs8GpVrj3GsKYAT3q2+ocwMxzg0ixO54Umo8xWuW7XS7qc4SJjnvjiti08E392f9Ww/CtLNq6FY39O+GM82PMVua6TTfhNx88efeqjRutRudtjotP+FsKYJhXI745robD4ewRKMxCuqNBWuYSqPY2Lbwbbxgful57Vow+G7ePB2CtY00jNzvoXI9IhX7qc9qmj09F5A5q+VW2J87kqWijqoJNSCIYP8A+um4i1FEQB4GPpS7Oc4xQINuGp2BQACjIHNDYwJ703gc5oE9RN2OtByOtDGkJnFIx9KN9RLcQmkLUJFWAHIoyQvNS+whAfSgniqYW1Ez3pCePekDQh+tJ2ptjsBGR9aOQKVxJCZ45puSenSi4kO/GkA55ov2GISQc0E45NFwGk5POaVTzSYegHk5pu71pWFuBHBOKacYqraAIFyPrQUx1qdNgvqMeJSc8Zppt1PJFHQBjWaN261Xl0yFxyi/lSsWmVJ/D9vIuDGAc9azrnwjBMuAoxUyhHdlKfVGZdeA4X/5Zqc+1Yl/8MoS5IhAPsK5qmHjJWZsqljBvfhpIpJUOuPyrD1L4eXgBKjJ965XhklZFqaepg3PhDULb/li3XBrPfTbm3k2vE4+o4rCS1ua8vVFmzti7Y2mpr2x8tNyjFOEbrUybs0ZgjZ85Jpv2ds9xSnJ2uynqS26OWwDVorLDgg/WqUrai0T1JlkDrgnJrPuYG35B4ocYtJsFo7jEXYeTVsSqUwaznTTKbchscZ6gGlWZ4nw+aqDtoLcc8gkxio8hTXPKC5nctLQc8oMeBVeKIl+TWsrbESJktueeRViBViaizYrlhtVitxhm5rJ1DxIAx2H9K7KVJ6MytqZkviCUvuDHmqlzqs8w5cjJ7V1WVrL+vxKT6IgVHnfua0bTSpJcAA80RfQzlJksmjSIeVNNks3iXngVFSTtZExlZ6kQQfjU0cQbivOmurOqKva5ctrKItk1LNZJ1FZuTtcpKzIlt8sB1q9FYRldzDmtaaUiKkXuJdCKAAjFRWzRySYPNEqd2KK0uT3UPloWUgiszO5unFNRtHUa1RftJ0iTkZ+tRuUeUsFxk1onGSsyHCzFKc5VcCp1tS4Bas5OysiluWYrRCvIo+xhTWMvMEncV7YMuMVUlsyD8oOKUJK9i5aakTxOnBzSLEfStbozceoxl2t3FKC2OtOPdiejuSLKyrg0eZlulXGb1CUL6mnZTR4CuRk0l1BnlG/CuiLTVzKzTKimVTyM1MiiQcrg1NSlF7FjGtQDkVZtlHQ81zTp2vcpRuSzWqkZ7+lUpLTa/NctPcVugwxBehpuzLdOapt7ovoOClB1pQzHrVQWmpKXRjlUEe9OKgCpktdB6kf2gVG8wNXKI3HUrlyTmk3c9eKqFiX5C+YpPNT2c6GTmrhHXQb2sWb+4McBweo4xWPaNunOa7os5rvoaxWmDA+tciT6HSlYPN2nBOKeLj5aG+iHuN8/n0NI7DHqTTcW0OxGxx0pAcmofmJq+wu/I6YoLZ6YqlFLUXKPjwecdKlb5lzU621AgkDE84qMkoCDVJbA2noOQ7qcSQO9XdX1FYaJCDxSu26m1oLUaBgetGAOvWo2ehSvuNzgepNHJ4q99Qt1YBucelOJyPWk+zGxpYg4FJuPU1WgmupHJKOQTioHuQvGaFG4O9yJrgvwMmrVlp01ywO04qoq+iKd0dLpmiBArEVtRWyrjjBrXldrGb1H7dtGO9apWRKY0t+dLnjmqBiE96jLZ6mknbQLMT7x56VG5WPJJoux26Gde6gqKcEe1c7d6h5khBIHvWbavZM6IQtHmMy8dlOQ3BqqkhLZyKG1Hc5m7vmJGY461GSQaxcruxolYQuTSqGBBNOV0tQvoSghhz1pQBmobKSJVRStQTD5unFTLXQlb6kJHeg8VVgtqJuYc0BznJpKyKWohkIOaPM3cmiybGlYMZGc05G9aFHQG3clVzilUux4BptX3FYuWum3F3wsZOa2bHwTe3QUhOtPR6DcbHQ6d8LpJsF0JJ/Kun0z4UpHg+TuGOeMmtIYdS1ZPOlodPp/wANIEwTCPyxXR2Hgi2iUZjH0x1rsVHTQxlU6GvB4at4lGEXjtir0Wkxr/CK1il1MnN3LMdiijARcVKIABjHFXoRdseIR2A/KnCMLUuV9BsXbznigDnJ4p7Awzg80m4fSgLAX45pM5xS6hZhRzQCYgBowaVxaCkY7032o3KQu3PekI9TQib6iEZ5pueapdgQEcZpMc80lcq4uKDyOaklvUb06UhB6nmq3KuAOTQelJrUXkJ0HWm4p9BeQZJo71HoO40gUD3NU9hLcUj3pCPekhtCZzQaGJLUafpRjFIAwaQ9afUEGabnLe9DQagWOc0Z9aBWE4FIT1pajDt1puMmmO4EDvTWjB60nsIb5f8A+qmtACOVH1xRZFddSGSxjdSCo+lVZ9EhcbSg5HpScV1GpGdc+EraUH5R+IzWLqXw/t5kP7sevSuarRTVrG0ajRgXPw6WMkpEVPXisTUvA1ywIXOP5VySw3I9DZSUtWc5deE7y0ckxFsegrMvLGaHgxsD7jFYVIO2oIrwI8JJK/U1JPcBozisk3y6jcb6laG4OcGrHl71zWl7h0IjbMx46VEQYZOQac/IldkX4hvQEA4qG9UEcHms0le43daFVXYcUqAucngUNX2K1WpOoQDmld0TkYraNG7uzNtlOXV0izgiqU+ulh8prrhRUVdkpXMy41OWRjzUJaSb3rVtNaFNJIclo+MkVbt9O83GetJPUxu7mrZ6fHGRuFbELRxLgCpcUggm9WJJcKeoqtdiOVDgAGs3yvYfJbUyls2LnFWPs+xcN+dZyp3LjPWwIWX7pyBTvOdjhjXJNON7m8Xd6CwsyzA9auTTNtAUYNaUZaMmfxalCYyE5Y5p0AK80m7O47pE8lwdmDzVYcNntSbu7guxbhKEfNUmyMnI6UuS2wSTLMSIR1FSuyovBBNK2gJa6jPPKjJNBuM9RU26sCVJCRxQ+epoUVuKUiCRVbk0KVx2pcjb1FcqzlZG+lRqCD0OK6UuVWM5PWw8xN1xToXC5DjmppxTuN36A65fKZqx+9C8GtleOlyJPuEZLfeq1CMdqXtPe1KlHmWg8IrMc4oMOw7lOKU5qTsSua4hkde2aazqw6VPsU4txKuV2GSccUKFA561l7MfM7AcelMzg5IqJOwNosRRCVcjFNlhKdeaybaGilNHgk5xUI5z7VacmCY1j9aRck80/Z6WJl2I5Qc9Kdbqcg4NbUrqOgndGjLGZbbrk471jLujlOPWu6npuRdm6xIGTULygdc1x+SN02QuSxzQPXNC0VhMlCg80hIHWhSlYfNciZtxpRmiUkU3bUTcSaRm285o1ewk2lYfFNnqamLg9OaqXYXkIX55qORgaPIl2vcSM88CnPnsaptdSluITxzSA470k10Ha+w4MPWoye+etTonqLVMcCueDmnbkHfmqb0sgGEr64qFp1jP3quK1Fd9BhvUHGeKrzakqDg5o5B2KE+rbu/WktknvHAXJroglFC21Op0LwxJId8qEj3rrbHSY4IxhQPwqFKN9CpRurlvyAg6YpG+UZp3uRZERYGmlyK1t0bJG7yTR+NO9gaGknPrS4qX3HfQguLpIVJyM1jX2r9lOacpJLUdOLbMHUdSOCSea5m/1NvM4asab1ub1JcsbCw6sZV2tUuT94d62rarmOZEiynrQzFuc1yOXVmlgD889alDjqTRUnroOMWLv4pQ+Oad03cSWool96CwPOaLhYbInGajV+op2uin5gTjvTSM81F0IY2T3pApJ7mrbt0C9yxFbySHCqTWhaeG725OUjJpOWti0rnQ6V8Ory7Kl0bHfiux0b4Vg4LxHH0rWNFt6kuSSOx0n4ZQQhcRAEdsV1Gn+CYIcfKARXUqSW6OaVW+iNe28OwwnhB05461owaWqYIUD8q2UUkYtlhLVBwAPyqRYQp6VUbiHiMUoT8qAtYUEdqUnK9aYw3ce9AOBng0mJsTcSaRiccinYe43PPIoAJpgxQMmjpUgABpMUbAgYntRyRQkG4o5HNJgUnpsOwHg0lArC8803H5immMUHjmkIHpSuKwYxSY29aQCbeOKCCeKaY7dBMelBHrQ9RICvrTdvpS1BjSKOtFw6ARj0pAKExC96Q+tJjuJjHWkHJoSGxDik/GgQfjSc9aGhgcU0g9aEJ6jTk80du1PzFvoJmkxzzmi4wz60jfWgNtQzkZNN78mh6BbS45T7UjUrh1DFJgE0XG0IyjOcUxogTzSbvqTfoRNaqx6fpVeXSoZM7kU+2OtS13LuZl34Wt7gE7APwrFv8AwHBJnKA/UVhUoqWhtGokc/qXwziZTiIgnuOK5TVvhzcwhvLzwe4zXFPDa6HSqqluctfeHdQsJCTCxA9BUcJnUEMjDHtWNknZjaXUHlZDyCDUQnE0gyKbd1Yix0NlChts47VSvLRASRWdWm9omak0yiUiQckUx7iFFOCOK2o0bK7BtvUy7zVApOw1Rl1N3By1dmkV5itoUpLlnOSc0xQXOPWm7vQexZt9MeQZ5rRtdLWIAuQKXMorYly7E8whUYGKjimWI5BrL2iWpUIt7kwuw3Q0v2va2c1ftVsHLZjjqBbvTfOcnNZz5Y6odl0FS5KnkU8zh0+lR7VEuGtyHdgkg4pBMd3zDirqRjUiEdC/ZqjndnkVc3xkYJGa4nTkjR6lOaNd3ByKRI+eelTzdy/Ud5SlsAVL9gDLknBrNNtX6hJ2IZLcxtjrTliY9a0i5W1E31LKQ4UHdTlt3bmm6j2GpA8D4xzSCEnvUczSdxS11JIgYutEjk9a0TvqiGVZpWbgHFQr5g9a1TUdx76FiK28zGKsrZhVyRWM6ktmLl7CYAzkVE6KT2FXBqKaDlBAAcg04u2fWlGo+ocuo5MntVmEGi3MwvYWbKHIqITPn2pJhcfHL60jEHniumErLUiS7EDHJzmmbiG4NaRcZaMWqHs/yjj8aiZlbgms5UE9iFK7sKs/l8BqlSdW+8c1EqbtZmtyK52sPlxVYp6Dis0rOwWaJPswbnvSfYmHQ81DnZ3aF1uw+yN3qRIehqk03oKUdbomkYLCaw94+08HvXfTXclruanmMM7qhYlmzniudrqdPoJ82Mik3c81L0EnqPSUA9aV2B6ChoXLZ3GggjOeaN2Tz0p8qtqTK4h+Y8UhYHg1SRd+o0jHPWnK5Hc0asNBwYnOTzSEccnNLqTLQWM7T1p5k9KpxuHQaTnmmqCx5qOVLUpaIVgVPqKaW9TTSuT6kLzGME1Tl1LZnmt40hq25Tm1s4xuwaqS6o7k89a3hBx1kJ2WqITeSPwGNIryzNtXJNW2tibrqbOj+GZ7yRWdCQeeld5onhGOABivPuKwm7K7NIxurnTwWIhXGB09KkKBa51q7oq3QhkGarScVqtCGQucGmMea2RCXUac/hSM4VeuKbYDTdxRrksKztQ16OBT8/4Zpuwoxd7I5291qW4kJVuPSqMlyw5ZqzqTVrHXFKK1M65ug+QxrFvbYlt4PWs4NxOapLUhgIVuetbFqwkTFdjXMrEdBzoUqJ5COc/hXDKLT1NIu4iSZPXmpd/vRJalRY9XJpxJocRtJMbk560obFOLsLyHpcnoaYxDHI4qnboHKwVc96mit3kOAOTUOLvsLyNSy8KXl6RtibH6V0uk/DCedlMiHaatQcrIE0lc7bRvhVHGoJhz68V2OlfD2GMYEa4+ldVOiuqM5VTo7DwnbxYzEMjvtrXg0aGMDjp3IrpULGDncux2ijbwvHtU4hHWqZL3HBABSkYAOeaVtSReooAIHWnYbfcKD0wT1pCSAEY9aDkjFJPUPUMeuaOKYCbec0H3ouO3cNvvQAM8UmxMCeaXaOtMAppzmkPcCPWloGwwSaCMdKbE7CHIIOc0devQdKGwBuvFIeOPWpuNCAgGlosAhpCM0WATkDmhTTsF7igDGTSbcmkhCMBz3NIDx0oAacHrSHrRbUOghPrQcdaLAJjJooYwJ9aTpTt1ENxk0hpXGGOKD70XJsJjj3oxgcmhbj9BjD86QA9c0dNRijHWm8+tTYV76CEZ70mKdx+QgFKV5z3psXkHvmmmpWwCFuaQEZ5zTQ+guaM4qeoBn0NJinYfQQr600xg5NR5gQPaq+euPSqs+kQzAgoP8abSa1K5nYyb7wfbTg4jBrm9S+HcLhiLcZPcLXPUw6kawqtaM5jUvhyygldw+vFcrqPhK9sJC6xFwPQVyPDqL0RqmmrmfNqklkPLZWU/TFZl14hfBHINdCgmyWu5kXGpzSNwag+1yOfmatbq2gNrYY7kjJJzUJVpDjms5dyUiS3sJXbO0kVu6Xo44MiU20yZTdjTnt4oE/d496zLpnbocAVyuT1RVPuyqI2J60ht3boK51Nydi5bksKbfvU2ZgucGrT7g1dkIcqc1PFdDoaTk27F26kpZWGajLYHWsndCIvNbdU8cwYckV0UZ9xOPUfHNJE2UPFPk1Fhy3WulpTITSCK/DH5mq0ku4cc1xTg4y1N+a8SW3BaQZNaKkY68VKj1IbuPW3EnOBil8tF6is2m7krcgkIB4NLFOVIpepeyLayoy5Y81CWTPrVOGmghHIPIpjHinSjrYRGI0JyadJCMAritJw0FqEUhi521bjuBIADUuGmpV1a6JFtFmyRiq9xp/lnrUxjZ6maqP4SNLYmrENhuOTSkNvUtC1iQYNRPtRvl4FVBaitoRyMuPWodu7pVeztLUForsYUw1WI4VZPmPNKbb+Eb2uQzRhTwRUYhLEEVKk46h0JFQ4wage1BOav2slqLk6ojmtSF+Wq2JFb5u1dEKnOrEpa3JI1ZyAxqVgsZy1TKGt0Vz9GOWVCcZqymKn2N0SPbbjNV2Zc8mlGi4sd9DN1O8CAhW+tZkEm6XPvXbBWjciT0NzO7rilAwOa5FY26gQhFMaMDkcmk1oKzTI2QjmkAY9s0ki7+7cTaydR1pQfxpy1Fe9hMc+lIw4Jp8zQ7a3EU9s08HFOXkJoM4O4nrTd4A5NF9BXuxpmUUfalBwSKqKYPzF+0xjvmozfqCcYq40m1qN9xragD1NQyX65JzW0KKexLKs995gI6VlzszE8nFatcoovUhMDNzSi2cnvUurrZib1L1ho0102FGAa7Lw94JLMsjqRWU6iSuaQjd6ndaZ4ehtkX5RWoloqdBiuNzbkdNtNAdcVWlxTjvoZO5VkbnioHye1b2bV2RYjaPuTUcrpGMsQK0j5kPyMy81yC3HDKcdq5vUPFrBjg1olruVCF3qY03iKeUn5mPpTVnmuOXJ5olU5Fa5rGGtiQsIly5rOvdR6hRyaw+L3iqjsjNM7M3LVKrb1weaJS1scjTvoRSWZHzKKktbkxMFYV00Z3RMY3djUEizJxiqUsZVulZV4tO5rBWdmQnKtUkcnrWTkyvQmSUA5JqXdnk9KXPpYNQ75owT2qUwVh6W5kPAyavWfh68vmxHEfxo1TK1Wp02j/Da7uWXzQcH0Fd3oPwsSMDMZJ69K2hCcndicktEdvpPw/igALRDjviumsvC8MWPkUfhXfGGhyOTNaDSo4eABj1q3FbKOMcD2q7djNkwhAA9KUKAaYkOxggilJHJoAReRzSHrR1DYXNIWzTsAvbNJk5pLuF+gqjNHQ8mhbhuBJxz1pACRmmhDqTGaTGwUUDOTk1LBISlycVXqNiA0ZBOc0tbgtg79cUuMd80AGM0dRQA05p2PWhsVhDxzTe9IBBSjnrTEtwIH1pKQ2IeetJjBpgKD60mR2pANyaXPHWhgN6nmkI5zQJXE79KDyaLjWrEzQx5ouVYSgj1ouKwmCKa3NCE9gwPWg1PULaCH2phznNUhoQhiaB1pvYm4nOaTGRk9KTGg6D1pDQ0J3uJ0oJFJsqwh9aQ0kC13ExSkUNg3YaRzSVNwFPrTSwDdzVK4WFzk5JpM80abBsJ3oyKHsMawB54zTDECDmptoJ3IJtPikBzGp98Vl3nha2uckooJqWilOxzOr/Da3us5hUg+1cLrvwfQuWh3oR27GsXS1udEamlmcdqHw21GzclYzIvtWf8A8ItPEf3kZHsRU3bZE9NUVb7R2h7HAqKxsQX+cD8afXUmMl3Ny1jtouqjipLvUo40xEBxUNpR16DVN6XMqS/eTqartcsa43O7N3FIRZsnOKsRzcZNZxVncrpYimy3NQt70TTSBEb+1Rc5zUwdkBOkxFSb9/U0NNg3cGjOKjBZDyKUW4iWi1Jknx2FLLIJF5ArqhLq9yHG+qK211JK5NPt714H+Yc1r8TC9lY1LXVA3XFX4rkOQd1c1ROD/r/MtK+pdF4VQdxTHn8z6mk7JkJWZAY2Zu9KFYNUuSuaSZMMkdKQwleaidR20Jt0I5GcdTQhYitVJbgPQHNSBOc5o5ne4PYjlfacE0xW2miUnuhJdC1bXLxHjvUs14X6jmiUuZb6lcq3IlnYN2FWIropyaztswaVtBzXSu3Wqs0oDZzmmk+hFncdGPMWkZGTnpQ273G3bcj3Ec1Yt33jmtErBciueuetMilI69KiUdNSoLTUsLiQcUjRnFZvYTlbQgZWXinJbhuwJqotrYT8hGg8s521XuEEvHSt1PozPlb1K0tlKg3K1R/aZ4xzmuuM09A00uA1R9pDVUm1YjOMZrSST1BrW6KEtz5rZPNOtuZQT60apXZEtrmwkhGM1KJM8E1wt6nTe41jjpxQJCKFLRxJcr6MUP8AjSiXnpScbsrpqKOTknrRhQ3HSrlFbINQIUDJ6moJJFXPNLl6EtsiaVF5yKabxRVqD6D1sMa4AGetQSXuMnPSrjTvuS5FR79mJ7VGbpyOtdEI2QpSQLPK3IpwaQnLVbYmxSrHpzmk8onqaj2ltAbG+QSeTSm2XHNZTqNopbgsRLABc1r6V4dlvmBKkisJbouMUjvPD/hNIEUuvP0rp7ayS2UYArKrLXQ2S1LifNzSlT3rNaM0S0IJF61WliJ57VotDBoqyRhckmqlxcxwgkmt4skyL/xBHEDtI44rm9V8RyTZAb8uKtST2KjA5251GSduv61TKPIee9Nzs7Fcy2RZtrLJ3HtVvKwISSOKx+OTuWtEZt7qPXBrLaZpWzWkkrWOWbuJgA+9SpgLUatXBXVmWoXwOe9NurcH5hVwlZ6C63KqzvE3LEe1Xbabz8An8665yUlZlX6ks8AVc8VWYbeQa4kuo0rq4iufxqZMkDrmk3bVAXLWxnuWARSc10WkeCL29ILRtjPPFEVd6FWS1eh3Wg/C5dyu8WT7iu80X4eRQhcxjj2rqpUXa7MqkzrLDwnDABmPkVtWukxxYIRfp611xhZHO53LsdqidqmEH+TVepDk7j9gAFAXByTTiK4pzSDGM+lDYIM0hbPNCRVgGe1L1osSJQB3p+YbIXFHapELggc0h69802+w7aC9R9aOlAhGzSk8UbjYnJNKoznmk9BgRSds0JhYQDjJox7UN6gK3XpQDikAm7B96TnPWhBew7vRn1oEMPJ46UE0+hVkJzmgc9qGJsOc0HrmgBDSAYpMewh45FJk/WmtiWLjuaTB61NxgeaaT701qAAEUcCkDAjNIwpdQuNODRnnPpQwQmST6UmKaB7iHuetJ3yaGA1gd2c0YNFxbATgZNJz1p2G9Rp560oyaGJMaRmkxx1NSHmIcntSHGKOowApDxQA3qM0ZPSna4CHr1oPWkO4hNJ+FTqtBXEJ560FsHNNbDE3ZNL0pvsNiYB5ozipu9hO7ANgfWgAE0nuKxG0YJweahmsIpQdyA/hQ9EUjOu/DtvPk+WBXPar4Dhl3MIV/Ks3DW5qpnIax8OMglVIrh9c8H32mktFEXAqOS2pVr6nOymeJ8SApg454qMy7mwx4rlm7ysa9B7wgDOeKgeIdqwas9Avcb5bDmjzGU1L0H1F8046imMc96zc3LUcbbCN0phpRVtWG4hz0pyMQ2TVIDSgZHT5utRTRDJ280KKkzN3RWkDJ1oV6u7SKjsSpz3pHtRJz0raE+5nJW1IDC8LfLmrUF8yEA9q2kva6MaZpRXwfAZqvwcjjmuOcXcpsnRDmh071nKm90K9xquU6mpPMGOaXKmVfQa4Vx2pqLilFWVha3JFAoeRVWqVxvYoySNI+F5NSRRMetaT0QLR6lgDFLjPvWF/e1GPWPAz1pDkdaL6kvyIZJCDjHWmlS3Oa0TkhNE8RaPvT5JSRzTUkK2hEHGOafGwHIpSugiuokjbxio9pBpcz2LTFWQoae1yTThroxNE0Y8wdKesLIM5FVtJkPsRyT8YNVDgsac5K+m447AZMe9L+7dTnFVGfM7sUo3dzPurWJs7TzWVPZsDntW8Jszd1uVWjYNyOKAzJzW6kPSxtF/n4o3e+a5JWTN4rQUHK5NKr54PSk99BNdRxYBc5qIuo/ixVxu02AfbEXrUb6lgkk8CrhFvcPNlWXVgDkHmqtxqpYdetaqHRiuUn1BjnBoS8YrzW2yJbY43khXGab+8kXPNW2lsGw9ImA5BqdYsis3Us9xJX3J4o1XtnFTFVx2yayqVOhaiyMjFROQvNYuo2U1Yj3EHmp7WCS5OApNVdvQpavU6jRfChlKu68ntXcaZosVoi/KP8KynLlLjHQ1EdYhtH6VaijMgya5Zy6myjoWUhCrQ6YFCd9wv0K0uFznrWdd3qR555FaxkZ8tzDv9WHzAE5rnNQvXdjhjTVdxNFT01MS7kc5yazp4mbmtYTbiJuxClm7Hp+NW4LIL1NS6ienUIR1uOmkjgXGaxr7UMsVzgVrTjbczrTeqM2RvM5JNIowKc2c6d1YUctnNSKpz71DdtCyeI88frVlWVhg0l5FSWhBcWYJ3g1VVvJfqa6acmzFXLS3RbgkmjY8rcA1U1fUqLuaGn+G7u+ZSsbbT3xXa6D8L5rjaZVJ5rBR5tOhbdj0Xw78L44AuYenqK7rSvBUVuBiJeBXXToqJzznzXOitNBhhxlB+FacVmseOAPoK609LHPKTbJ1jwORxT9mDmhDuOxj8KRjimTYUN70m40xeYm40ucUmNiY7ZpSKLgApcYPWk2MCN3Tp3oxkc8Ci4mIOKcWo3DcM570vBpDADC+9Ny2eRTVmJgASacB3ouDEPT3oA7k0AhCSRSZI64ot0GLnIo+7zSa6ALmk980gvcbjFB6U7jDp2oouD1CkJ5pCEJzRzigdxDxzQCT1oAM8ZpC2KYtxOopNvqaWw2BAB4NITikDEDc0NyOaaZNxOS3U4oPHINA2GeM96Q0lqMTrRjPehoBPakIxQtAsGM01jz1oAOvvTT70EiE80Fu1EmMTvSGk2AhYCmls1Vuo0A5pDxxUsVxB160jGne7HewimkJx3o6g/Ibnmk3daQmJnPNIWz3okNBnjNN3/nRuULu74o3ZFTJdRNgDnmms3NGzsFwDE9TRuFDQXDcPWnbgOKTE77CEgnnvTTGH64pq27DoQzWSSjlevasjUvC9vd7soOfXmo6lqbucdrvwutroE+QCezAc15/rfwsurZy9uTgc4bqa56tO2pvCfMrM5690C8sgVmjYY9uPzrLkt5EbBGK4ZNp6mmyDy9oJqF8Gk/eQXdxpQGk2Y6c1m1bQExDnvSYGKm2g1uPih3kCpJLbYKqN2ge9mRbinep4pAepNXfSwpakksI27utVGXBpddQi31FBxUySY9qqUuwNXJoSrn5sU6a2RlyowfWtoPW5lNWZWXcsnOcVtWUgWMHd+FbzpuUbod7luO6JPHApZrvFcs5JLlZaSexD5hc5p5c45rGDVvMH2IjKc4BqQOyAHJpti6DluM9aUnf1NEouwatDBGqtmrERHfrVPWOoRdxzKD3oUe9c89EX5Eg5HWo3HPtV76k2dxhiJOaciAU5SE1qSqABUcpBFZq72E0yuzcUwPj1razuG2xKjZHNBcCptrcSBXUtyetSlUPOa2hFNA2TQSCNcZqU3Kgdfxo6XHa5Vm2n5s1UllAORWVm5XBJsill454zTVlPanbQfS41gSckHmgqnVxQp2sK3NoQT6Y0x3IpxVeXSZFUkiuqlU1a/r8iJx5f6/4JPkYJzSbsVlbuaR2EZ9pzmopLpFPWnTWtwa1K8upAAjcMVUk1TnrmuyEEIiXUCx5pk1yWGQauMVfQTWiKpkdyad5ZcAUTtHYmzsPWzYnnpVlbMBRmsp1H0GotokWBQRkVN5SgcVk5PmRXJpoJt75oAqXuVy2QuSKcHxyaJMcRC/rUZO9sVlYGi7Y6W9y4BHU12WieHY4hkqCfet/hV2VFXep1ltaxwIMCphKc8VyTnc1hEsQRhmBbmrsfyfSsrXepqmTCYKuSRVee9SPuM0paIW5kXeosScHk81i3dw7k5PAqHULhEybgkmqU0G7k1LmhtWVyo9iWJyKi/s0FvmFXCq9kZtDXt44DyRVC/vo414OK6qFO75mKUrKxz19qRdyM8VR5Y5Ndbl2OOTvuOGe1KDxzWfXUSXQeqAnNSqpA96mbWzGkA3Dk05Zj+NEEuhdu5YjcuOabPYtLyi5PtW0HfTsS4rcv6T4Svb912xnB+teh+G/hW+VaVcnryK2jroxJWPS9A+HUduFyg/IV2em+FobcLiNeMV0xpxSMZTN230xIgOn0q4kCqfu4zWtjG9yYR4B9KUcd6aJHepppOe/NGwrhyOM0E560MbA9aQZI6UdAF7/Sl6ChsdhD696UdcmhjFpOvfmlYVwxjrQT+VG4biDJNOwDgGgNhCcHpS5Jot1AQkjoaAT1p9AsLxmgnNJiQnfmjvQPqHSk24OaaGOGenam55pDBhnFAAPejUnpcO9GcdaTHYQjnrQSBxQK4mcmkzRYYdTR9aYhKT60tyriMcGjqOlIW+oi5zkUm496fULADSHmgYnGcZ6UZGaBWAGkJBPSnYEGe1HX61KBsTPPNN5J9qAQHJNBJJ60xBnvTS2TUrcGNBPegniqYhgag9eDRsVZBkdzTWNLUXUTPtSE980DELc56UbmotoSNJ5pGyBUplITcDmmZ5ouDQjHjrikLd80DsIWpu8nNP1BLoNL4pPN4waW6GLvprS4pILCefxSCfNOVmFhPtAB54prXKjIzUsQguV7mlNypxg5ovqOw9Z1YcGnCf3oQh3ncUu9SfenYd10GnYxPAqrcaVBcjDIpzU7bjMLVPBVvcKRsUk+1cNr3wujYu0Uflt6gVzVaKlqbU6nQ4jV/BN7YqzCMuo9Bk1y1zZSwuQykEdiK86cJ03Y3st0RFWxzSYJHNDt1JAx5GaWOAk57U9GrCZYiXa3Ip9w2V96atytMfLzalNxzk4FIvBHNRZ2uh2LPmMVqMru6jmne4nsMZRn6Ugzn60KOmpKJgaPOYZGaunoG+rJI8Pz1q9aqrkZIrpjUJlHsaflIsWaou43nnis50ue7CDsPDADINNkLEdeK5vg6GiXUYgKndVvzlKDND94GtAXDdKdg9Kafcm1lYQqepNSQnb16UOSCKHO+ehpQ4HWocBtDt4zSF+aTTSsNvUPMA601mx0NaOHUXmw831oJ3Hmp0Sugk7vQZLGTzURQUKRL7IcgOeBxTZlyM5p9LojbYiVmHepkct1qi7DmfapzUUczSNxVdNWVTi5FswMUyage3IXOKxctdCZaOxUkUNIATmrkVmrJnFVOd7G3K3EjaIKeQaqSSp9qVGI5NOEdGzOkrSVzuItGgOnLLn5iuf8a5zV2jijcIM8EVjhpqTdzfER2OUkv0XoagfVl6Zya9SEHJ6nO1dEMurE8A1VmvHetFBJkyaK58xz3oEbBue9U520QS30Jltm4OamW2JxkiodW2w42JkgRe1PEaoelYObbuxa3JPlH1pRkjORWd2ty/Mese7kdaGBA5NF23Yp6PQEXI60uAOlRzNsVug0tio2lArWKXLcdraiIrSnCitXT9HaUhmqrBKSR1Gn2cdvjgD6VsRXaRDAqZXNKepZivmc8Hir0EucZNYyj3NdC7HOqjk8fzps+pIg4b9ax1Tuw6lCXWj0BzVZ71nJJYjPvmsa09Fb+vxN4QSI3nD96qTkE1zSbTKtYgNuHGc0wWg69anmv0Jt3K8qqhPQVmXd/FHkZAxXZQouTQmc7qutbcgY6nnNc7c37zN96vWirRSOGpJSZXHzHk1KAvaou7si2lxrMAaReeTU6tDSuTKwFO37RnPWk03uWtrCrJvNTQQtMfl5qkuViaudHong68v2U7Dg+1eieGvhdna8sZOeta04tu4nNLQ9F0L4ew24XEf6V2On+HobdBtQfiK7oQOWc+psQ2CxgHaKtJFt4rexluyQIOwp2B35paiAsKbuFAagWyeTRyT1qmFluxQOaXv71K1ExB1zmkORjk0DADilC4psoU8UnOakncU89DTSMDBppiFU+tHWmO4u7ijOOakaQNjNHOetCfRgwyMdelJuwME0EtBuoDgepp9B2aQpOaAec1ILUCRnJo5z7U0OwnQ5NGRmhABNJnJ4FCEgP1pOh5pN9BgSDQcAUDEHqaM980J6ktCZxQeO9APYbliSetH407hYQn3oyCKTGJuprHIzR1ATntS7sU2NjTzyKOnWgQE56UfWpTATNJk4yTTHYQ/Wl3ZHNJiEzTd3PBp2ZK3FLU38akfQQetIW560wGk80bgDVMLaajSS1NJI6mh7WGG7vSZB61ACEgHrTWejUXUbuIGc0jNnvRoNoYX4prSHuaSGkNL8U3zR34pbh1G+dnuKYZgCcnpQNLXQY9wO5FRm9Ve4p3BojbUk5G7pUT6mmOGqZPsOzWpXbWFDHJ4HvVeTXQM4bj2rN1FsPkexXfX+Dlhz6VG2vMTwxP1rGeIjtc1VOzI21188NQmuP2Y571zyxT+RTpalmPWyOdxz9amTWmPFaRxFyZUugHXnXqfxpw8Q/wAW/H4VTxSS0JVF7D11535B61bttXDNktWKxetivZaGlFeRyjnHNOktYp1OQDmu2Ek1dGFmjK1Dwzb3IIKLz6VxviD4a292hPkjk8MOtKrBTVuptCbRwGtfDa4syzQgsB2NctfaLPavteNlI9a8urTlSfvam2+qKDQsvUGnqfLHTrTi7xBNPQVW3GphAWGSamzbsU9FZFO5TYTUKvtPJq17ugo7Exm3YxViIKV5IzRotCXtoRPHgkjpTDgVlK/NoJ6igcdaULV36Da0HqOanSUpyOCKa5lawrIlN87DG6k4IJzzXVTq6cpM1roNSZlPWpxdrgA06tLmSaHGRNHMj9OlP2BjwRiuF+5I0a0uWYhjvTyuelSxLfUVIi3Wgx46GpTFJa2GNGVPWlWMkCrUrhsiWOAscc0PbFetTzNNNkshkU/lUYDZ5zWyldWCO2pIF7nrQT3rKLZVugBsjmmSYPNTfUeg6B1HB5on+YYArS3UnYrxr8xyKsBNq5pyvfQT1IJ+FOal0+1km+dRxUTk+W50UPiuy9K7Iu1uSKqzyfJzUQ1FWtfQg0iEXepJGRnJ6V011pAs1UsMAmsqs7SSbNow5qZkapEkSZHGa5K7nC3GQehruoxlyM44+7LU7LSPERm0rynIOFK5zWRqM28tyT1rkpQcZNWOyq+ZI4J5JM8k0xQxJyK9yUk9UcCloTxWzMeatJZ8cisak7D6DvsoHOKPKGckVjdyKHhe9LjilqJ6CbjQWJoeg3qKNx5p+9h2qX7xXQd5zJzimtOWOSKKa1u2N90C3OOopGuuelUoK+427DGuN3epIYGnYYFCj0QpPU29O0sIAzCthNkC/St1G6uTbmAX4HBap47sEBtw9KmSvqjZaIsR6tHH8rOKsf8ACSxIOoxWc46DWuwxvFAkzggAVC2svKcls1yzi46X/r7zpjG2o0X57mnC+JPXNcskara5It0zVPtLjJrnm7LVjS6hvCDmq11qKRjrTo03UkOWiOe1bXFUE7sEH1rlNR1oyscE8172HgqaucFap2MeS4aUnJqIZzVN82pyt3HEletLvPpUpIrpoPVSRk0u1gM1Dum7FJaajh05FSwW0kzAKpOelKLb3ZTR0ei+B77UJF/dkA16V4W+Fe0KXjye+a1pwc9ROXKj0zQ/AkVuFPljI9q67T9AjgwduMe1ejCCRxzqNmzDaJGOBip1iArVLoZtskVR6Gg4FF9bBZsM8UhPIoBhnn1NBPc0hCYOM0DJ9qpA/MdkjrQefwo8xCHIFKc4zS5hoOgyKC2enalYYc9TSc5p6PQB2cConkA5osSRtOR1pwnHXIzTsPQQzdKUTjdnNOyAUzA85o80ZHPXpzU3HsHmA9DSGShA2xVf3zTt9JgxQxPelDUMNhSaN1INwBPrSZOaADPfvRnjJpiEySaQnikMTJoJ5obGFBbvQIZyTk0pOaGMaWPagmmAZyPemlsUxCc9aQHcO9Sxh1600tzQK4F9vejORnFEl1Gg3UhbB61JLdhC/Gc0m7PWmkHmHWmljQFwLdqQHH1phYXd6mkLDFKwCbuKbnHemhXDcfSmMfmz1pdbFgXppY8k0CvqMLk9TTfMINACNJzTGY0m0lca0GmTHemtLilYYzzuM8VE02O/NGwdbET3YHeoJNQVTnNLZD5StLqyKfvc/WqsutgZGc/Ws5TXcpQfUqS66ecfTrVeXWWI4PNYTrJLRmkYFd9UlbvjPamm9kY/e49M1yvEX0NeRXGvMx5zULue3Wsp19DRRI8t3NOG4kHNcc5X1QxQpNIQw6Ck5Ma3LECsx5rRhiBXkVdObb3JkJPANtVxEQfeprSfQFfqTxxFcVYQHOT2qIT1sBetp2jAOTxWna6hu6k/SvQw1blZz1I9S+s0cg9z2pJIFkHPNene+xjsUbzRILlcMg+veuW1nwNBdI37pWH61MoqW5UZs4HXPhtJEzNAuMdFI4NcZqnh67spCssTKB3xxXBVw7g9DZa7FFLfYw3CrBdAmAazg+rLs7FG9VWGapNGRyaJp7olN7MaH2mgXG04oje5duxbilEi80NHzxVVF2IbuNAcHnihjxk1KjdDfkNSUg5PNTCTIrRq4r3QoU9RUm9lFStAXdj0wRk8mmsmTk9K6ISfcybGmV4RkHAoXUGz3qnTU9S4yL1tfAj5j+tXoLtWPJrjlBqTNW7l1QDHuFQlyOazcbaGfMMeTdTRIy9aduUpFmKc7c1J5vme1JK6YPRXI/LG/wBaUoPQU7EEMiY+tEUJdqqKuU31GXUDR88mqxJHBqHFqTiS3dKxIo44pxcYNVzacoJEZcL261KJ+AMURbKtdFe5JPPaum8M2iyWZbHSscQ3GDsdGH1ZV1ePyZ2HTFZbtlelZYe7SuLEfFqHhyYRa9HuA68V6L4nEP2ESgDqCCPpWGJivao6aNvZ2PNtcv8A5ig596524QsCwU171OFoaHkyleRa0u4MakdKuzNuUnHaudp6na5e6jlja5OTThbhTmujnRzMlRcGpA2Cayd2i/IcvINIVA61Mfi0JsJt9OKOT3q76agMKZ5zQFx1ocnYq2g9QMU7Axmod0xpdGMdtx9qacU2iUraERPuajOWPFaLcrqWLTT5Jn3EGt6zto7cZYVoo2M5Nt6Fn+0I4+ARVW51UdmrXVaG0Icpnz6pzlWJqMa1Ko+8R9KqLt/X/BNOW+41tUkc8sfzp63z9d1Yzkle39fiJSJo75l/iq3Dqp4Ga5qqUl/X+ZvHXcvQ3Rb+Lk1ZjlZiBuribuzaKLsLY6npUsuoiMYyKydLmloUZl1rIGSX+lc/qniDb0Y59jXp4egoq5yV6mjSOZvtUedzgnmqLSMTya6ZSVrHC1cVefxpxYKeTWLetkVFC8MeaVBluTQ3cqxKARzUqRvJwqk0k7o03NzRPBt5qbgiN8H2zXpvhT4VRrteWIseOo4qoUnNhKXKj1DQfAUFug+RV/DmuvsNBhtwPkAr0oQsrHFN3dzWis1QfKMe9WljC4zzWpk9RwAHag5xT9RCbj60o5OabHsKfl7039alC3F3AUhOe9MBelLn8aAYZNJmgGKenNAJqUgE5FHBbNUFgOBjnNG6kFhHk4zVSeXHNAFCW/IOCfyqBtVVDksMD3p3SdmNp9CGTxAijmTI+tZ1x4whgJzIPrmoqVFG7NY0+ZhH47tpDt80dO561ZHi2AAESDH1rKOIi9hypNaMlh8W27DPmL9DUv8Awk0G3Icc+9WqqepPs2h0XiW3kbiQDt1q5DrMMpwHH51ftE0KUHEn/tBcZLCpYrxHPDDJ96E0S7EyzrjqD+NOEoxyRmh66ghVlDdxTg2T1oDQCR3NIc0XDZCc55ozmgBpoJ5pthYXNMJNIYFuMn8KaWJoQgzQWPehiEzjkmmluaLgGexNJuPrSBoTfTW55pisJuzS7jQ9UVshC2TQ3Jx1pIVhu7J60En1oe4mxFfHBNGfehja1ELCm5xmhMYb/Wk3e9MkQP70hcUrg0xrSH1prSepoW9yrDWlz6Uhlx3qWx2I2lpplpgMabINRvcEd+tIfKRNdAc5qvLfKDwRzR11Go9yCTU1AzvwapXGtKjZ3ZrJzvdD5dShc64c8HiqE+sO54Y/nWc6qRpGmyA3sjd6jMsj9TmuGrXVzdQshMt60oyTXLOo3oi+UeqE0/ZWcrpXC+ohRs0CI5zUOXUBwiJ7VPFaFsZqQdydbMAc0n2QA03EOYfFb7asqpxSjpqhMPKZqX7IQcmibcgvrYUQYNSpEc9KyimtAZZWE4zTvKZRntXRTk07mba2JYZ3Rsk8fStC3vs4DH8a9CjXfUylFdC0sglHXFJJGDxXfB31RjbXQqXOmxTryo/KsHVPB8F2jYjU59RVTSa1KjNo4PxD8NRuLxKYzz90cVwur+F73TmO6JmA7jvXm1qPI+ZHRF8yMv7E5PzKR9aguLcKCKxk20ybsoyxACoQnPIpJ3RTdy1bDA61ZVlTGTV3b0JHAJIeKjlgOOlO9hkIXB5qQD0q4h5Dw2Bk0pYP1NFhbIAxQZzxT1mD01qhWurhLhveq0kHORWkJ2JlcERo/mp6Xjo2a1tzO5SsjUh1cqmMnNSx3u9uTXNUp9y2luWwRgEVG2WPpXNZ3BWTHKxA61NCcjNTd3bL3RIud3JqwkZf0qZy6sjl6siuIthqOKQITzV0p9xuF3cJpSy89KpuN5xitLptvqS1qCxkDk8VG+Vakl1BdhMfNnNNEm1qqG4xZGLD2rr/AAaFNo49ef1rlxmsLI6cIrNlHxD8t4+7k/TmskgMDU0V7qQ8SlzXMtpjZXqSDOQc12Goa0b3Siu7J2jFOrFOauVSnZXOHvNzS/jTXGIT9K9hStCzOCcbzuU7Nv3+CeM1umPdCcenWuVttaHRbRHM8mgH1qk1axikKPejmk9wEDEH0o3HqaaVirJDhJxSZ5pE6vYdnvSnkZpX0uXfqhoYDvmgtkdcVSb6iI2b3qNnHeqSb0BO7BYmlYYzitKz04febFWovQllxrmK2XHFZ13q2Wwp4rohBIqCSKrXh7tzUEt078E8Gk5X3NtBiyY70pOTmocuoNdRpfB560puCVNTZ7sLa3HQzsT1q/bSHjJqKiRrE1LaXHJq9HcYGSa4pR1N79h7agwG0HaPWs+/1ZY1JZsV04els2ZValtjnNQ14tkBjisea7aY5JrrneOiPPlJ3I+fxpwTj1rC6C9xwBFAw1T6FqI7Zk8U9IzkACnfoWlrY2tI8O3eqOFSNsHvivS/CvwoLeW0yZ6HBFKK97lRTXJuer+HfAcNmqjysfQV2VjoMUIGEPH4Zr0aULI46k2a1vZqnQcVaWMCulK5g3qSAEcUH9aBiKeeaD14NAmJRnmmAH6mloQASD2oX6VIAx55oye5FNbCQvPrRmkPQO1H40XDcQtzSE8VVkDBnFRmTvnpSQkRSTBVJJzWZd3uwE5wDQnpqPfY53UdXEefnx7Cudv/ABGUyBIPxrGpNR3N6cGzmtT8YPFnEmfxrEl8Qy3rHDH65rwcbi5axi9D18PhrLmZB9rnVsrI351Y/wCEgu0j2+a3HfNeVSxU4OyZ1TpQlrYgj8UXcRI89hmp4/Gl3u2GQ/XNdMMVU2TMnho7k8fiu8VwwmJHp61o6f49uoJgZGyuelbQx1SnZmU8KpROmPxBD2ufMBOPWq9l8SgJSpkIweoNelVxsYOL7nFHCtpnR2XxBhlQfvB/Wra+OYeP3uAfzrqWKi1oYOg0KvjmIHHmc1paR4uhvZQhkGc9zTjio35ROi1G50ccokQMCDn05qQPXWYbDdwJoD4NAhGbIzTd1AIA2Rmmlsmk2AE+9Jk5pgJnBzmjPvRfQBpJ9aM0nsKwmTSZoAaTigtQwQgbvSbifWkVbUTePWjd70xeY3POaCfencQ0uAKQvigdu4bu5ppakIC3vxTTJihsa13GiTI5PNNaTB60uoWGmXtTWk4oegNEfm01pqllETz4NRyXYwabYFaW/wBoxu5qpNquO9Q/IpRsUrjWACfmFZ9xrPOd1ZSqamkYNlOXVnbPJ5qq9zJJnBNc063LqbKCQio55yaeIs8muCrVb1RdiaO3J61ILbvWVnLcLiNABzSBcHpU9bDT7j1Uk9KnjgLUO7G1YnFp8uTTTbAGofSxCZLFag9amEQT61Si0wu7h5ZPFPWAkc02rrQfMPWDHNSrFms+W24rksUQzyKe0Y9KvyF1IzD371LFFnrWY29CxHGAOaeV4qyGtSJ07YpmCprSMrbCsSxXLIeRmrsN6DjJP4iu6jiOhlKL6FlWVxkU1kB+ld8WmY21K89pHKMMoOfasTUvCtvdK2U6/jRNX3LjNrQ4fX/hyrFnjQo2a8/1zwveae7FoyV9VrzcRRad4nSpc2xzs9syE5XmqrxEGsIsT0BXK8U5mJFWrhbqSW8pU8mre8OvNXoyWiJoqYRinF2G7DSc4zQCc5pt2FGzHF8jmmFtpyDxVReg2SJPup5bvSSIlEPN3DGKYYg5ziuiNoiUdCCXzIzgZNOgnlVgScVokpISka9nqCkAOa0o8MmRXDWp2NRjFQcU5X4OOKwknyj9RpnYHrUiagYxyRU8vM1cqMtQe9aWlgTzDT5EnZCdt0XRbHbzUbW0WeorRU7Mnm6lW4dUbatV5It3zZ5qmrbCTIzkdT0pmFz15rOLY/Qc6ll4NdX4HjaSOQc4C1hi5e5ZHZhNZNEfiaJo7khurKD+FYBYrxTw65oKTFiVYyNWDK+/BwO9aNhdiS0Ayelazp7SuRQlzJoqTQ7pCfemTwbYjn0rqcvdM3o7GMjGO6/GulsJRJbkEdqdN3VmW/4djlixoV6OUxi9AD5pwPvSlvYr1FyGHWmlRjiktBMTB6mkB2mpab0C9gLj3oMpx1rWMU9wbG+YB1pDMCcUSXYaWmo3fuJBNS29k0zZ9aqPkLo2asNolsoZse+agu9USLIRq6FEIrmMS61BpWJLVVaY9Sea0crGqj0QonIxSpKSe9Sy7aXJQxwTmhJT0JrJWdxqNwdzxk0wdetCWg3HQmhIQ5OKuWsylxkYrKavqXTi9TZt4iQD1BqSSVYVyT0rKMOaQTkkZN/rXlsRnOKwb/VnnY+ldiilscU5czM/e0h609VAPIqJy+8ze5MmD3p+3AyKxaW5rHQbgmnrHxzSuhpalyy0ua+kCRoxz6Cu/wDCPwxlumV7mMkdQMcGhRb2NdLansfhf4cwWkaYjUfQV3Om6BFbIMJj8K78PS12OWpU7m3b2gjxwPYVaWJVxiuvlXQ5W7jx9KATVKyE1qKGAPWk35osFgzz1pc0h7oMd80ZPpQJC5FBINGomN3U0yDOc0DF8wHmlBHJPNMGOVu9BYkUrA0KW4FNLcdKQr2GF+aCwA61S2BkTyAc5qtLdBO9DAzrrUh1JrndX1sKrZOPxqJPlNIq5xesa4xLMrEk965m81ZpQea8fG4lJOx6uHodTCvbhpGOSTVaG5MBPOK+flPmloetFWVixBqayNtZqmlk3JkHispR94UlYzLqUoSd1Vo7lmbmuiEbajjqzWs7jcACeaskBuprGpO43CzI5Z5I0IDcVUF06HduJNa+1bihRilc1dMv5GA+Y1rrcynB3HisvbtadP68zOVNIeL5weWPFaOlavJaTrIGPUV1UqlmpnNOmnGx6Z4c8XxzQqryAn0JrqIL5Z03hhg9q+roV41YJ9TxatL2crEqyqe9LuyeTxWr7mIM2KTNFwAHrSbsUbjQmaTfzTJ8xGOTSZoGNJ560pcgUPsAhekJ7ZpWFcaWP1pM0DbE3UFuOtFgG5NN3EDOafqF7BvJ4zQXpNA/IYXyaQvzTSBDRJikaTPTikx21Gl8H1prSHrml6iGNLjv1pjTelC7jtcYZ8VG9yB3qX3HuQvdbe5qvJfKv8VN7FWKk+qbed1UJ9V6gt196zcuxUYlGfVieh/WqU2ovIfvGsJ1UjWMCrJcyP1P60wIz8k1w1a6i9DVRJUjwcmpVizXJKbluVYlSPualSMZpX1sJkyqBzSlsVTdgURjKWNOjtyTnrWM7t6FbFqKzzgkVbS2q1FpXIkxTD2xQluSaTV7BfQmS3ApWgqmmK/cckANSiAjkc0RSsK48QZNO+z4H0pOL2HfUVYsHrS+WT9Kjla3He4ohH40uzFOxI5RinFhmnuF7sUKHNO+z88gUctxNsPsozk9KFt9p4x9atPl1Qr3LChhye1SKxPWu/D1rPUxlEcxG3NRugYZx7V3eZCK89skq4cAg9axtT8NwXIbCDkdxTaUtxxbRwXiT4cRzEyRgo59BXA6p4QurNmBjYgdwM15teiou6R0NqSMmTRZVXO089OKhe0aLIYEGudS0HZpXZWb5G5qeKYEcmtKaAkLj1qBjzwap3uQnYOtG4UuVu5XUbvyaVsNVWsiOtxETDVNjIxVXdim9Q4XrUkVN7XFbuOIQtUsUET5BxWkHZeZnLYhuIlh5U1JBqpRdvarlTci4Xvdk8d8JGxnrVyOTK9RXHKL2NJasQkE8imNhjwKxemqJlcdHGwbPUVft5lhwcVtCKbuxWurImuL7cnynFZs07k5zxVuVthvTQYsm/qac7lV4rGN9mU1pcrMxzzUiYAzTktNCHsK8mF6V2nwy2PMynnII9v88VwY2UVTbZ24Ne+P8dW4jvQcABlHT1rlBGGb2p4OfNSSLxmjM3xAyRwgHvVTR7gGIjPTpXfKLUEzmw2tzTjAYZAyaLiENGfWlNXjdE1LpnL3uYrnn1rX0q63AKT1rSiuxUXeBhk005xTZi7IFGKUH14qW76jTvuGOc5pS5Wm5X0GnoJ5hPBpu/mrutkTqG7vSMQByabdtirMjYjsaYELHjqaa3KS0NG008kBnHFXvPjtV47VvCnpdmbTehmahq2/IU9ax5bgtklq0UrG0I6EBcnkmgn2FEmmy4rsPQc5NS8CspPXQqwbuaGYk1HNbctbjk5607zAD2pxvITFV2YhR3q9aQFTueiyehE5WNCTWo7VME5IHSsTUNdabI3cVXs1B3RlKbehlSztJkE8VGEJ60Sk3sYN3YuAOlLnis03fUY0OQ39anViRQ0jSO2pYghaVgFXJNdX4a8A3urOrPGVQkHkdazsaJKLueweD/hhFbKhaLLDuRXpWi+F47RRhBx7V10Kb6mFWeh0lrYpGAF7VejhCDpyK7locstWSBcj6UE4q/Im+oobHfmkJ55oauMTGSfSg9KV+gCjFLzmgLDiaacnmhCDJFDMBzS6DInfioHnx35piGC6+b7wqRLkE53CmmDJRcADkYoa5UDO7gUW1FdAt3GRkOD+NNe4HXIxTS1uxPQhkvEX+LmojqKg5LUDK82qKqlsisbUNbjVT834560roqKOb1LxMiqfn6e9cfrXihXLAPmuGtWUTtpUW2c5Jq/nuQWzVSS4w/tXzuKqc8rXPZpQUUVbiQE8c1VuImIJUVzRgou7NG2tyoodWzyMVbjviFwWq3HmTuU3cqXbu2eDiktxxmlJ2VkOOm5YjuWiYGtCK781Mk81zyWhs1pcZJMxOCaFi3r0pX90ylpqjR0uPYwzW4iApWNRcy0MpS1uV5PlfipI5Svfmrp1bKzIauWrbVZ7Rw0chGK6/wAPePWjASaQhsV6OBxbo1Ensc2Iw/tFdHTWvjKN3A8z71dDZanFdR5DAn619NRxCqyaR5VSi4FwSZoya6Uc7QZOM0Fs85pB5Cbu5pM0XCwZ96aWIouKwmcUm73pjsN3HNBbPFANCZPrQGwPeh7AkML0m7uTS2CwjOT3ppcH1xTAQPj2pC4NJhYZ5nPFNL+9O7GxrSYppk96V+rBq41pfWmNJjnNTcFoRtN19qhe52jrS3GlqVbnUBGM5rnb/wATi3lKlwOamU1FXZSgNh8RpOn38n61FNrGf4qh1U9UXGDKcupFwcGqzXLue9c1SrY6YxtqKoY9aUJ6iuGpV53oPqOWHcelSCIiubVvUoekeetTrCeOKS01Yr6j1j9qljiyaqKdxMsC3+Wk+yjvSkmxJ2JUtxUsdsM0RiDkW0hAHIpxAFa30sTqwwD0pyrUWHckBGPSmPx3p3shBGcdTVhCCeTSvoOxOhXNOOD2qfMnrcTAzSiP8qtJbsWouAMk1E5HrUyavYpXuM8wf3qVCGbmsyrFqIYGalBGOa10sQBPFIo55NJ7itoSqwHelwM8VtFrdES2GMSOtRvKFGc16FOpdWM7FWa8SM81Ul1ZeRnmtZTSBRbIWb7SMFAc+1VJvC8d6SzJ9a5p1FLQpXRWn8AQeSz+WAMdhXlnjnQDp93lYzs5yQOKzq04xhdI0jI5Ga3Rx2BqEWxHesKcrIGOEeMgnNRONhrS99SRobJpzjildt6FJaDCmeaACB1qk7NoGOx/e609X2npmqi7qxKsBPOaPMx0NF+hchd+TxT1lK03JqxNkEku8c9KgkiyMitlU90mSvqJGWj5zVy2vtp5puPMhxu9jRimWUZ9acXAPSuHlvJxG07gszNgAYpzLJuz2rafZCVlqJuIHLc01eepqHctO4xwd3yipEBYYqNtwHJbbjlqJohGMVPM7aBqytL93rXYfCydE1NVbkFtpOfXNcOYJ+waSO3CfxDY+I0QjaOXjgkVwMl4tuCWzWmXwtFWLxi1Oe1u/wDtA44FQ6NOACMkmvXre6rI5aCtPQ3bZzuFW2O5a56myURVnqc1rqbJCcc0/QpB5qg9M1WHvFf1/wAAinrFlHjvQF96drq5FtLhjHemmpUVexXQCcCjJ6Gny9BqwwikxiqSsKzZE0hU8UhY446mrst0V1JILd5cGtWCzSFQzjn3rWNNS1E3roNudSjhUqpwaxLvUJJGPJFb2sXThrqU3ckEk81Fk96hWNGrCgH0pQuOpNTvqXFEi8UpNTJpalLcRWx9KVieoqGle7CTEMpPFSwwlyMjAq+XsRoi6qx267iRVe41QYIRsVoqdtWckpNszppXkyd1QFS3WlJtshvqOVCq5xQSQvNY31Kiuo3fkc0D5j1NHLq2U0r2JEj3dOTWtpGg3WqTKkcZ574pWui4xPU/BnwsyyvcRkngjivY/DvgyC0jTMY4HpWlGi27iqVLI7Gy0pIgMLxitCO3CjgYr0IRsccncmSMDgdqkJOBWi8zNiFu5NGaADvSc55pXGOHNGaN2CEznmlyMdaBMTJ65pC5Hei4wL/iahkmOcE9KBFWe4IHU4FUZrwA9RVOw0tClNqQTqwOKSPWwpxux+NYtsrkYk3iaOBCzOOPWuW1z4jR2qttlyBU1q/s48zNaOHc3Yw7L4xqLkK8mFz1zxXa2Xji31G3EiSDJGc+tYYfGqs3Hqa4jCunqU9R8YLbcl8Ed+Oawbj4kRK+DLj+tXWrqOrM6eHcloRyePUlTiQHHvXPar45OWG4/nXJPFp3szqp4XWxzt54me8BBck1galqUuSQ3WvNqVnKWp6dKiobkNhqT7/m6Gr81yrrwea4asZOWjOmy6CRPvIGavJGNgBxWNV23IlsRPZBwSKzZ7do5O4FEZPYiMtbFmKFJUw3WoZ7fyTlapK+xo3ZlZ5vm5FT20hY03FRRopaGhDDvIyK04rQbe1c3K5GNSRLbxCN89KuG62DHWtlFNWMHqyEzb2OKkQnGTXHU92VzpS0Hct2pYt6nI4NKNT3gskSJqUttKGDNke9dR4c8azRTIkjnaT1zXo4TGSpzujLE4dThdHpmj61HeQqdwyRk1qCQN3r7GnNTXMj5ycbMXOetITVIzuIWpNxB5qgsJuB9qbuNKw7Bn86aGweaoLaClu9NJNJDcdBpJprOc0CQhejd61LGNLc800tTuA0tmml+OtDF6jWb3qNn96SdxoYz56mo2mx3pX7j3I3mz3qGS5IGQ1O6W41qVpLzGTnkVRm1PB6jP1qL+ZS7MzbzUhgjNcf4jP2sEqcn+dctZqUTSEbsp6c1xGQDnitZDI4G4nmuL2vLG1zeMUkTou0c1IqE/SsJVOZ6l2Vi3BbFuasC0A681EY3u0Zti+SFpClDWgIdHF3qwqA1KV9x+Yu0ZqWIgVSshEofJ5PFKGBPUVSSBEiOoPWpUlX1p6WuIc10nTdTDcg96iUlcaj1Y8TUhuTnrSckiuUQ3BprTsx5qXK7BRQ9ZWqWOcg8mhO4W7FiOYnpmpfMc8nNU3oRYPtG0c0fawO/NK9rgo2InvKhe996hMqwz7VnviporgDHNTexTRbjusryaeLj3rZvUyeg4XAz96jz1PQ0k9bhYcs2TwelSLcYB5quYkBMpzk0x4RKpwea0jNx2J5dTNurKSRjtBqsNJcH5s/zNXOcm7ItWRq2mnFflNbNpp4A+YD296KUn1MZ6FqbT18s5HauI8T+FINQMgMatk9CK9CFpRIgzyvxL8NZLdmktMrjkoeRXD3thcWEzRzxlCPXvXHVpcmqOjfUr9BnNRyoSKSWhKIVQg81KGAFF7obdxM59qCvrVctmMTA9eaR2285ojcmwgcHvSnk9aVmlcb2GEtnIpVck/Ma0aTSRJIW4zR5mRRGCsHKBBYUwRkciqhK10ws+hYtrhozya1YJoZcAnmr9knqJX2L8FtEec4pl7MsXA5qPZ2TuSr3sZTzkvzUkcuRWEka2VizHIgHzUwybXyOlJpOJSVtyT7Qe1RSOzHLGkkuWwPYqzS7fpW34Fu/K1aM7jwwPWsayUqTub4dtSTOs8dyM9omB91ienYj/8AXXnGqBjGTms8vanG6WqN8Wu5zd3uwcmptBbMpzmvSq3UWzjofEdAGx9009J2UcmuZPTU0qxRja3IXy1Q6LIDdJk/xDrW9GS6GVFLoQFgOvWgP35rP0JirLUTfmkL8daLNFdBu8Ueb+tUiegu7NRswHU0PVWCKfQiwZDVm2s/MbLdK1jH7I2zSV4bVOSBWde6ySdqGumlDlTCML7mc85ckk5JqJwDyTWc5Ns6YxsQnOeelBIxnmh+RdriBsMDTicnNLZXBoXccZpNwHJyalu7KYgkBpY1aUihuzM5aluKBU+Z+aW4vkiQ7OTWkY29455ttmbPqDSkjNVzJk5JzVyqGGqY8yE96d5hXnvWLeiKSWzHLPng013yaTSTuNLohqL81TQWskr4Ucmk5GqSOw8LeBLjU5kLo2089K9r8F/DmGzVW8sZHfFOEOd3ByS0PTdK8PxW6jCYI9sVuW9oqDgV6FOHKcc53ehbSMD2qUADrWpmxC2KCx6UgG9e9IVOc1SaQmOHTNOJ9aW4wD8elAIoATd2prHAzQIb5oHWozJjmi1hoY1wenGT3qtPcBc4P5UrhZGbd3oH8VY97qYXPzClJ2KSuYl3rW3OXx+NY934j8oE78jr1rlq1OT3johC5y+ueO2VGVJGzjsa4bUfEM15KS0rH8a8fEVfaSsz2sLRsuZmeb0h9wbJNdH4c8WTWJ2FztPaufD1OWe50V6Sqwsaer+JpZY8q5APvXPPfs77ixzXRXr8+rOShSUUWre/bH3qiu5DK31riT2aN1HW5AiMp61FcRFzznFZ1JO9y5NWI0iEXPep4vnPOabbtzAtS3Em05xVxZcjBrCeurEy1bsCMHvVe+jQHdtGajl05jPW420gDjpTLq2AOGFEW46xNb6mfNp7E7h61LBCExk1UZO2pKlrY07UDjNXxcBajW+pLTbsO+0bhnpQsoY8mnGSSsgcCzBGrHOKsGIGoqRjYSk07DkiwMnrUsUW7kCuPlvLU0cuoy5sgwJ70y2gMTVpGLiDqaWOo0HxDNZsqs2BXeaP4jjnjClxk19RlWKvHkkeRi6Gt0bkVwsg4bk0/cM817yZ5ohbBzSFiaGDYhNBOKOgWsNHHWkLc5o6gIz4pobjrTWwDWbJ60jHufypbDGbu5pA/vSYhCwJznmms+KQDPM601pOM5psdiJpc1G8veh6DIWnxk/rUL3GO+alj6lWe828ZqlPqAA7ZqW9ClHqZtzqnUlsfSsm71fB+8fzrFvU2ULmdNqbyZAJqJEeY5bvXDVqo3UEkXIrVV5xVhIq4ZvXQCQR561YiQZFHULstIwUUrzYq1LlRNrjC+acG45qea71KsOEoFOE/vVXRNncQ3IHSm/asng1nKSTKsO89iKVZnJwKFV1HoWI97DrUgVzSc2KyHrA71Olg5qdRN9CZLFlGaeLFmIyMZotfUXNd6ko01scigaax6jA9abStcV+g5dNYngGpl0k5xj8jQoPdsHLXQvQaWAQDnNTvpY8s5AFdMIozc9TLvLbym71QkjYH3rnmmnY0T01K8oc55qIK2cc1k+bZFX6EgQ96kQEGrafUVyyr7V5NRyXHvW0noJXuNF0c9eKcL3aetZruW0noA1IActilGog96OexPIPhvgT1q9b3oYgZFaRd2RIuKyueoJqaKFJDkgZ9aG3e6MmXYbdcg46VoW8SgjIraldET1JrtAsJPXiucvlJkJzXoUnomQtjMudPiuEIZcmuT8ReCbfUI2UxBvwrWUFK6ZalZnmfiL4f3WngvApZF7d65CWOS3kKyKykdiMGuGdPlehpvsIUDAetMaMA5zWaG13Gk0u7indi1AnPFIY9wyaSbtoFrDDFt700E5wPzq1LoKMrj92T1pSgoemwJiN8tN8wLQm0U9dWRvcnOO1Kt2TQ1dA7dBwufWpIbnDZzzW1Op0J5eqLseqvHwGJBqUXnnckmqnLmX9f5hbqNJJ7UAOOma5XHlKTS1F8w45o8/B55oinsNWZIt0Ohp+9W79aFGyBxZBMoAyOaveEJiusJx1NctX3qckb0H7yPSPG1kG0xpBngKevqK8vvV3o4I55GKyy92VjsxmiujEe1DFt1Raeotr0YHGa9KrNO8TzqWkzcZcANSsAVrni7wub1u5k6sp2Gs6wk23KADowzWlKy1Rz017w7OaXd601bYOocHmo2YA5p2b0H5DXGen41Ex2A5aqT0FvoIJflAzSrukPFUotLUezsWoYkQbmpLnUo4BhTXRTjZ3YmnJ2MyfUmnJG6oN57nirlJ3sbxhZWHg8UueOTUNdTRbDXww9qYG4xjFZ6hYaTt7cUhl49KqMbq5QnmE09QXAA60OIMtQWny5fGaleaG3QmpjSbZy1JPZGfcaizHCkYqm8xfqa2k9LGCQ0DIpRxUNahy3F3c0u6pZSWo4KTyKcEJ+tJu9kWkaGmaTPfShVjJz7Zr07wX8NGkeOSaPdk5OBSjHmehTlY9m8N+C4baNSIlGPau107S0iVePrxXfSpqKOepI1ooVXpmpkXFbpaHP1HHr1pC1O1gEzzRmhq7AMnOaX9anQAJNGe9UgFyAKaWHrQtwY1nOOoqMyDuaHtoGxDJOQOOarvcc9ee9S9EFrlea82nlqz7rUQmcMPfJob0Ha5h3+sKgOXFc1qOuhc5esJ1FubUoO9jm9T8QlQSH/CuZ1LXy+fmryq9ZvRHq0KF9WctqF7JIxO7is0zHPPNcWtz1IJbIcsrCrlm5LbsnNYzXL7y3LtoakLmVcMc1XlRkfPalJ3Zi4amjpaCTqBVm4smBzjisqk+WREtHYh8pl69KeIVkXms5NPVhIrXUGw9OKLUDdzRGbcbDgaEYxxjIqR1UDtmpkk9EQ1qLBKQ1S3A85cVSSaszN73IoC0JO7NQ3l2WOcc1vTo3WpXPqRrdZXBxUe/cc0SpqOqCO5es2BGSass4xXLO5d9RsZ3d8VMAV71y3cXqbXWxIl00fQ1It6+c5NbRmm7MUqatcmGpYHIq3aagjDkgUckW73/AK+8zcH0LqzpKvBBzUZG05okly8yMVo7MVZCtW7PWJrSQFXOB70oVZQ95FSgpKzOq0nxpgKJH/76611Vh4ghugPm6+9fWYHHKtBX3PFrUHBs1EmDrkHNLu9a9J6nIxN3NIX560/IY3fzyaaW5poEhC1IWwOtJ+QrDS2aaWzR5DY1iTzTC1JjQhkqNpMnii1gaGF/Wo2l4+tK9xoieXAqCS4xQ3djRWkueM54qpLegfxZNQ3qNR11M251DGfmrKu9UIzzWUnY1jG+hkXWqFsjdnPeqDSyTN1JrlqVLI6oxtuWra3OOeprShh6YrzJvmkEiyqBRzS5Hak31JSHopJ6VMoNSrtjsOBNI5wKbuK3Yar45JpWl681UZdxtEb3GBnNN+1FvepnNLRFWuKhdzk1bt7ctjNZJ82pLZfjswRzU8dgPSrcepCk9i1BYnPSrC2Gecc1XLcUnYsQ2WO3WrsVsMelVFdyXIl+yg9ufWhbXnmm0r6E3uSi3HftT1tc9KTs3qDkTJa8cgZqdLVM1pHR6E+ZN5OOe1RzuFQitVZK5OlzGvsM2cDmqDqDXPJqTZtEgkiHPFQGIKcmoaKVwApSAvWqSCxEzH1qByaJO+hSIyxqJ5HNRq9CrkbB8U0yuOM1C00BO4+KVwetWorl1I56VpBvXUmSW5OmpuDy1aFprbKRkk1orPVmco6aGzY62jnkjn3resbyGbB3KM1utdjnl7pZvXUwEgjkVzV2Qzk5rvpqy0M0QEcVFJjHOPyrRD6GfeaTBeAgqOetcX4n+GUF/ukEQDdmArKbTNVK2h5j4h8IXmiSNuRnjB4ZRmsFg2TkYrks022W5CBO5NNY4NTG7BADuFPBqoq2gWHbA61E0GPeh2TJ21I2wnuaPNOKdtbjWohk4561E53DrRa+w+hCRinrjHApaho9BrZXk0iyHPvTVlqK9iRZKs28p6E1XMxFoXLRc4yPepY9QSRsdK1cY1I36obV7iTbQNwYEVD5gY9ayb5dLAthhnAzk4p9vOznGam72LV+pfS2ZxkmpdFza6ohzgZrmxCvBl4d++eq+IT9p0DJ4zCp/IYry25TDktj1wK5culuehitijJCMkhayJh5d0PrXc1aZ5kX79jcVA0Ktknikf5UpQV00dFZMztQUtGSTWPZkC5AOB81VCKUrHPB6kjA+tMY046vQnyGkk0m4VpJ6NFKIwnBJzUbneeelVdNIpb3FSHPJzipfOSJT7V0WbQruRRudUL5APHpVGSUyHOa05nFGlrbjV69afkg8c1PM2y4vXUniBGSaXcp5zWcr3ujTfYazkH2prygDI60tWCiQmQnk5pOXORVt8uxLfQuWVmZMFxgVofZoYE3Gs/e5iJS5dijdagEyFPNZc07SsTmulS5Tnk2R4+tKBnFZOTJtcf0PGKUc1nqWxAOcU5V70OTT1HsSwgk4Heui8PeFbjVZk/dtg9OOtTJPmVi4b3Z7F4L+HCQCN2j+bjnHWvWNC8Nx2ygbR+VdtGnojnqPW501rZLGBgYq9HHjrXWlpY5m7skByPSgsR3oYCbuM0hbNO4uomfelB9TSbY2OyOtBbFCQg355NIX/P0oSBdxGfNRtJjNAyJ5sdiaryXGM80bgVJ7rA61Smvcc8UXWxUUZt1qiqD81YOoa0Ap+YcVlJtGkY6nM6rrTHOGrmb3UWdicn65rzcRWsj0MPTuzNnczg881TfSzMK8CvWle6PTi+VGTf6PIjHANZNxaSQ8svetaVe7s/6/A0jLUjXJ4NXrL5cDsac5dDXY1I3VFBzSO/mcZrO+pGrdyzYSvE4x0FdDCyXEQyRmuas09WKpHqipc2w3ZFNS3Kjcam90ZPzI7hVYYNVljCNkVcY21CKLkJLDpTjGSc9qaaWom7Mkjixzip44880nLlZD2JXtVZM96oz2O7JxmtYVTJ7FGW0MZNJBEc4xWk5trQuD3LUQMZwelTg5rCT0Lv2JYRg5IqVvauaS00NVJkeCTzUyL61i9Hc1uRTofU0kUzLwKSqu2hUF3L1lfNGcMc1pC6UruyK6OdSjYxrU9boQXK9yKa0y5+9Skrx0MkmILsoeDVy08TTWLhg5PqM1tTqOnJPsOVFTjY7vwt4wS+jXc+CeCCa7CC5SVAVOc19hhKyrU1JHg4ilySsO35o3c11dznsNL/rTNxFC2H5AW+tITRewmNZ6YXpX6DSGl6Y0g7mpfcEhjP71GZMd6G2OxG0vfNQyTAd6m9hrcrSXGMnNVZZ885xSbHsUri6xnms26vsA81EpFqPcxr7U8dWrDu9SLEgNXJUqdzrpU9RbOF7ggtnFaMVsqV59Sqm7GknroWkQAfSp1fArJeZDHhmNWYIdxpb6CloWhAAOlBTBrRQ5SLiMAByaryyAUpu0blJO5Wa4CmmPcFulc6mjTlBA0lWYLfkUrc2rBvoX4oB6VdgiAwatOysjNl+BBmrsMa9xmtVqrklyOFcCp1iU8/pWiempm2PWLB6VZhi9qh3CxP5B60qwCmJNDxb9OKlW3OOaSS0BsCgU8Ub9taJ9ybEctzjvzVWaQsP1pNt6D2M25fFVCe5NZNu9jRbEE9wsfLGqpulc5BzSk7LUqO4qsTzStuPOaIu4bDOD1pjjNJ3iwtqRstM8qnFjEaOoWQZ6UpaghVUe1P6UldsBFGKlDle9XFvqIVbt4zwxFXbbxFLbEE8itefqiZQurMvf8JnkYZhx6mmjxBHKQSR+ddtLERas2YeycWMk1hOi4+uai/tIynK96c8VFKyY4wkWLaR2PJ61fBDIQR1FYRqBJGNrfhmHUY2zGDkdMV5d4p+GhhkeW1Qgk5KV0OKnG6CMujOBv8AS7iwcpKpUg9KoSKR96sFK2g76iRYFSbhjpVp3epbiSJj1pG4PWpbfMTZ9RhhBOc80Nb4XiqcugpK2pC8WOoqJkNTzNbA+4wgikB2mnvqNIfnf1IpkkY65qUluDQ1SM81OjbTmm/MFsTed5i81HIhxuU1cZWegWtqV2uXHBbgVLFLk8nit5ttERk9h7unTNXLJQCD61zNs0s76mmZjEoJHFQ2t6p1GPOMbuayqxvTZVF++mevTKLrwqh+9+7cH6V5PdPtlOW/iP8AOvKyxpO3U9WstCGQjZweTWFqAInBPXNets7s8q1pHR2R32KNgZIqJsAHPWlQacpHTVT5blK85Q8VgP8AJcZ6HNVze8csHqWGIB560xipHNPl0UkN+ZBISp69aiaQp061W4LUbuLjB6ml4VcsRWlNa2toU+wye/WMY61mz3jSdDXYlyaMSTWjId+TS5+XJOKylqXYVTnvU0YPXGaT0NYIlII560xnxjipV2y0rjPM555pjyZ4GabXvBLTUdBavMe+DWnb6ekK/Pyae+hhOdlYJ76O3XCkZFZlzqEsp5bj0rRpJ6nPKStoVGLOc0BMnipcuhI+jHOazsWkBXmjv0p3uyxQBj3qeC3eZwqjJNS3rqOx3Hg74fy6jKryqdvXGK9u8H+A47NEzGBjHUda1ow5ndk1G1oei6XoscCjAAI9q27e3C9DxXoQVjkk7lpV2/jTt1U0QhN2DSMaVtRiFzUe7HShIYFznJpfMzTJl5DlelDhjjPIobBCM+KYZhnrzSQLzGPNg8moZJ+ODinbqNFeS4yMhsVUnu8k9Km+oWM+6vsZ5rIvNT2g8496iWmpojA1LWMZw1ctqmtZOCc1zVZvodVGCbMS5v2Yk7j0rPe93Ng/rXhYqpzN6nq0adgQ7uRSJdNFJya859TpsnoJe3iFQePrWZceTOh6ZpRpW1sJR0ujFmh8uTI6Zqa3bZ1rrjJM0buiwXJ6GrdjGZGGTSkrbkp9DUW0VRkdasxh4lyOB9a5KisxKXcabnBw1WI5FfpyKiouqKcRlxDuG4CqfluWrSnJW1MVoy/ZwY6mrEsG0ZFU4pq6MnqyEEqeatRkFaxl7xVtAMhFIG3HBrWm09zOUdCK4hBHQVWWAJyBV310Ji9BWjxz3qaGDcM9axm29jRbE4t2HNKI2B+apjG24c4uAetIflauea97U6YbEvleaueKY9oQOBUqCSY1OzIxC0ZyKeJmXgnmnFvcuT5hGmYDOaja8YHrWyqaIUYIabtiM80j3JK8mnOTepSiWtK1qeymDRsevOa9K8MeNVlhUSSDPoTXsZXiFTlyt6M8zMMM2uaJ1dtrkVxjDVfWQMM5619Kpp7HiuL6gT3JpC2TmjXcQm7NMZ8UJiGlzTS/qaOo0RO/PWmM/vigZG0lRNLUpgQSTYqvLP71NxlSa5xVK5usDBNKT0KSuZl1e9cGse91HGTmuec7HTGN9DBvtRZm61VtFe5nGeeeK8+tM7IxSR0tpb+XGBxVhVwc1ySj0ML6jwctgVZihJPPSlZNiZZWFQKmjYJVKPLuRuSedxwajefFbSd9gSK8l2B3qlPd5OM1y1XpY1jHuRrukNWIrcnrXPDUblbQuQW+O1XobU+mfeuhq6M72LSW7elWobcnqKGrIVy9BEcdKtxKc1adkQWVFWosHrWl01YTRYRAxqzHHtweKLoUiZY91SLERzih6szHbMUFsVSstGFiCR8E1XeXrSuVFEDsByTUUkvBwelK+gyhKNxNQSoQDWa3uUYOqpMzHaKhsLaRT85Ycd6zak5FJo1UXAFI471rHYEhvbmmlcnNS3rqNbhtGKjcgdKaXUCF2JNRn1NZylZjQh60oyTzTi3uBKBgc1G7VpclbkEsoUdaqyXPvWcqnK7GiiQNPg5zk0+CR3bgmuV1XcuUdLmhBAz4LVq2dqeCa6YavQ53LQ17e1HBq8luCOvWu2nexzyZOIQBiszVNPiuMhhz34rrpP3iHe5xPibwRBqCtlV3HoRXl3iPwZcaY5YIXT+8KeIp6cyNkubRnJzQvC3I4pqvuPU1grtoRMjADrUqKG71V9Ljd7EgiyKXyvk5PNZ+gX7kbwjHJFQSqBwSKe7K6FZ1ANMcDFJSuKMWRjjnNOL7hiqfvDvd6jCo65pyH3NUr2swWmrAyY6npSi5ycURv0BO4yRfMB4/GoGZ4x1ropyWzMZK2rG/aSfrVy01EoetX7NWuy0+5oNq4ePBIqGzkLXqNnvWE6a5WjSmlzqx7ZpDmfwvFjG35l69civMtUh2XTkkfeJ4rxMHHkm2z060ny2Kjtlc5xWNqPLZr1JpXR58tzodEQy6cp7c1FIGVyOnNc1KTdRo6Zq6sUboE5rBuvln59a3hZyscXWyFFwD160ySTeODj2raKsyrETOTwTSiIueTTcfe0BtW0HFViGc1nXd4VJA710QjoKO+pnvKzNkn60gcn6VpJ3NBpbJwKcoJqbWKehPGnPSrEeVqJPozWK0A8ZO7rULNiheRV1qNjieY8CtC30vgM/Sre9jBytcsvNFap2HtWfdaoxGFPFW4qKuc+71KDyM5JJ5pNpNZyfNqRboOC8Uh4qFrqUhRmjcTSsty9GLjI60AfjSWmhUUXLHTJb2QJGpJNeneCPhu8rpLMmT9KIrmdhylyo9q8NeDYrONP3fSu4sNNWID5entXo04WRyTnc044QoFSoNmc/hW0TK/UXfg0F85ptEsTfjqaTdSvfUY0tg1GWxRcLDGl75FJ53vT6DsKJ6UzEc55osG2xG0/HJpjXI5xQ+wmiCS6Pciq8l7gYz0pSZSWhUnvepBH51m3V/tHJ60r2KsZN3qgXJLfmawNS1bBJL/lWE5m0InK6trZ5AbrWBdaiJDnNebXm0tD0qNHZkSXW4fMaTyhOw2/rXi1XfU74xaVy/DaFVHNVru2A5J5rn5n0KUmzLvYX25B4rKkeWI/eNdUJ8ysyOdJ2Ym/fyRzUkMLSNgDrSlDlbY1JtmzZaK0nJHFaEWjvFyvWuWpiNbDlKzFkt5omqZAzLznNTKqpILq1yvPA2c96IC6MMmjVoanoaUTqy81BLtR8giocGkjNpkkUoI61PFJuODV8zSsKw94AeaFG3iooy11M5XGyexqHzcNgmtmlDVBurDmfNMA3NzUc9mJQHNDkjFW4Lcis3PoU1ZallYdo5oEG45xWkfe1IvrcbJbY5AqB4+OaxqQ6s3pzHWzbTVvIZea5nNvctx1uRmNcHFUp4juJApyfUuDs9RhGF5qjO3zcHvWi03NY6sdCC49qLjKLnNN3eiBu0rCQ/Moyasx3s1sdyORjpit6UuVoJJS0ZvaF4quopkEkhIJHPevUtE1lLqBSWBOPWvpMvxLk3GTPAxtDkleJqG4BHXNAnFeyefbqwMvHXFNL5o02FsNaQmmGT1qdAXciaSomlOaLlEckuagebil6h0K0k5HWqk1wc5ziovYpIo3FyBnmsy7vPm5PFQ27GkEZN5e4yN3NYt5es2RmuOrJLc6qauUUje4kxzW9pmnCFQSOa82a52bVJWVjUAxQxob0sYk1vDubNaEKYFKKdtSWyQkAe9ROxzTl2Ehhk21BLcYB70nIpIpSzFqbHG0jVxyk5aG225pW1t0rQitAxrdU+xi31LsVnz0q/BbY64rSCa0ZDZYFuM9KsxW4FEo3JuTCEjmpUjPWsn2C5ajgOOakWMjnNaxVldk82pPGSvUVahOae2wtS3GvNWAvc81oSxj5PNV5Sc0SBblaViWqvJnBPpWZSK5LMDmo3TJ5ptdR7kbR1BOowc0NaC1TM+WMHORmo1jC9BWe5b1FPA64pm73p3aGgIyelBHtUoEyKQkVAxJPNU3oUIEJpTHip5U1cNxPK4zSiOhJ7iuDjA61UmcL9abY4opTzk1Slc5zXn4ip/KbRI03M3tWtp9uQAT3rOleT1HVlpY2reHcQe/0rVtIeOR1r1qSSsjjmzTiAUCrCMBXWjKw8SYqvcReYD61cJWehNmijPCOjDjrWNqmhxXkZ+Qc9ciu+DvuJN3PMvGPw5OHltE2nrtA4NecXuk3FjOY5UZWHYiuWrDld0b3VuYg2OO1AmZDk1mpLYiDUizbXQbqanf5hkdKW2qH1sV3Y9s1WkJPU8iqb103KbIX6ZNR7t3FQluNu6sIQV5pPMAPvVK7epO4pxjPSkDYppXBO4jrmo3G0ZOaL62QkKkuOtE8wERNWo3Ymu5nGT5sinCVs8GuuN3oxaD0EoOc1o6fclZUBHGaJtW1Loyu0z27whKZvDeFzhH3fnXn+txGO+mXnhz/OvmsLb2zR60/hTM/BIrP1GHIJxXq1bHnz+LQ6zwOiT6e6NjKnPSoNVgEF06jgDpXBFr21jrkvdMe5HBrn9TBSTd79a6o25jz5Wiym3JOCaakbk4GTmuqm9bMpMuRWJwGcGkuZkt1PODWlNN3bFbmMe5vXdutZ8rlmJJJNbR00NFZDOG6mjPOBQ0xtDgnvzU0UfNS27XNFduxYVcDJNI0p78VF02VexHlpW4zVu205pSC/Aq5aJWIctWXhDFarnjj1qtcapgbUNWmrXZzz7mXNctKxLE1GSTyaiUr6MhsUdad0OSanbQIpgzUmCTnNTsi42Fx70pX3qV5leg5VJ6Vp6NoU+pzqqIxHfAzSe9iuZbnr3gb4cKgWSSM7jjIIr2Hw/wCGY7VF+Tke1dlCnbcxqzuzrLOyWMYA/KtCNNgArst0OabuPz+dKSDzmqehAjH2phfPSlcfqITnrzSM5FDtcNyMvzyTmonl55pg+5E8vPWojPz1qWx7iC4x3pftPq1PmQJEUt4F/i5PpVWW/wAc5xU8y3ZXKVJ9TAzh+tUZ9VXruqLvcrl0M6fWAA2Disq61UknJ/Ws6lS25cYOTMHVdXIUgFua5m91t3JAbP415uKxLi1Zno0KF0ZFw010SQTUItpAPmFefWxF5WO2EUtB8dqSatwR+WcmvPlJydja5cludkXHNZEl9vcqxJop9R01qNkYOnrWXdRBmwK1pvqRUT5ixY2AkTHFamn6ascoOAazqzbegRXKb0cAjUEYqQMMc1x6X1GtRkgUgmqxIzjNJQbkTYa2AKo3HynK5rqW9kKO5A146DGahe9c961XvLU3SRLa35zgnrW1Ynz8HNTVgmiakbal6RWjSqjSE9a57o5lrqQvMQcZ4qNz3q97F2tqRC7IODyKkSUlt1KcUnuaJaXLUM5Y1eim2gZqGlF6kTiWlYMvrUigEV0Qa6HNJO4SKGXBNU5xsWol70XcuD1Kay7Hq5FKJF+9iuJq+p1y01HxrufGakmt6qajoZuZn3abFOOtY2WaYgmq5dLs6KLuW4ht7054/NODzURlYuS1uOSERgZpXCHvTtL4jPmLFhGN2RxXS6Prr6e6hpDj0ruw9dwmpXOPEw500dnp3iGOeNQXHPtVz+2I92C45r7CFVSipHhyhrYmW/Rv4hil+3KR98fnV86JcbvYPtiufvj86DPnnNO66CsNaXHeonl75pDSIJJjnqKryzYyc1Mu6HaxUluKpT3AOcmp1KM25uuT2rIvLwrnL1z1JdDeETGvLz0bk1Whge5kzniuGtO51RXKrm1Y6esQBbrWgmBwK5obGcm2SDgVJDHuOetE9iLuxdjQKOakDDPFEdhDuSKY3HWhrcRXmk25zVCaUk8GsKkrK5rBdRI4i5yav28IAHrWNON3ccpF+BQDWlbx5rqg9bGUmXYlxVqLOetbOyJsWkXJ5qykZzyKz6kk6wZqeK2Gam2oaMtpF04oaM56U32JHLDk5qaOMg0op7DZajNTK3FaszsRSy7e9V5H96V1YoiKbupprQ8VKiHMV5VC1A3rS8irkTHAOTVOZwTVT7DKz4NRMwAqEtRkEj56UxD61LetirWRMEzjFKy7RVpaCexXl571FjJrNvUa0HqtO25qgF8s0FMU1sK5WnYgGs2dixNYVnbYtEDRkioHhJrltoWpaj7W2y4znFb1lanbnFVhoa6k1Ga1rb4NaMSY5r04RtqczeupNuIoEvzc1pOVthIlSXNWFQN15p03oD3GSWof3qs9lgV0xm0ZmffafHIMEHBrivE/giC+RmCDJzhttdNudWYRZ5hrvha40hzvjLIDjcK527s8Ekda85wcGaNKKuigQ8b9+KtwXxXg8itlZ6MtRTRJuVhuB61XkOMiocbyItYjNu8lR+V5bc1bn0KWorkEVXePac5oTFew0sexo3kdadidBwlx1qOe4XacdaSKa6me942fSmm4eXgmuuMUlZGUmSQWry81bjtdp55xTc/vBPoTBBnkVaihUENjmsKjbNKa965678PZxJok0ZfLbFP61y3iqDytTnOe9eFhWvbSR7E7OmjGVQf4uarX9pJ5RbBxXsyi7XPNruzuaPgCZxdvFk/MK1tft3ScvIMEjFeY+VV9TrjrTOcu+DxWDrSfJkV3KPLLU4Km5WtbN5hwDWnb2awLlgK64xV9QbvoRXl6sYIBwaw7pjKSTVOaj8JdOLW5TkSoGiGaXPbU0lHTQjZAPWgKB2rSLuiWmh8ajuOam3AVLuzaKe4m9u1Sw2Mk5B5oVlqKTsaUNlHAAzdRTbnUVhXCYrSMVvI55Sb0Mu5vnl7kVWyTSfYzFAox9axk9dQim2L1HShu3FPoXYMUvGOalbXHFCg8etGCe1O5V+hveHvDtxqs6qsbYPfHBr2rwN8P1tURniJPHanCHMyZXier6JoaW6qAgGPauihtwAMDpXpwgonLUldl2MccDp+tSK2evWtDIQmkzznNS2OKG789abuwc5pg7iGSmGXjmiwiu8vPNQyTBeppya6D12RVluucZwDVd7whvvEgVLsVy2IXvgDnOBUMmoj+9U8xfKQTasB3AFZ9xrAx94fnUSceoRi+hmXetqo+/WPd+Igpxn9a5qmJUN2dUKDkY974iJ53HHpUMOt+d1J/GvOxGLVrpnbTw7Kmp3ysh5rnfN3z57ZrzKlTmdzrp07G1BBGsQbHUU14UftXLVTfvCV07lZk8mQ88Gkkb5cilTTZq+5HCTMSrN36UlxpO4bh1pVJW1X9fmEpWaKX2d4ztINRTW3PStIyvZjbLelwF5AtdFBpTAA81nOb2JqSs0TbSnDU1yOtQ+VxT6k8wyQZXiqErFWqZ+76jvdjVm3nmlkRX61Wt7ofLYpXNn3WqxtGz901dOf2WUnfclt7MlxkVsWsflAHJrSUtNAcr6Gikokjw2KpXCFWOOaxjGN22YtWkUZCScmoHuCDhiapRs7mlrjGkyc1LbzgnB5qd3c0VP3S6p28iphc7cc0N810ZuJdtLnfgk1e8xSM5q4W6GFSLTGNKPWoZn3ggU2rpolIzLjMbGltJWLfWuaLSdjsjrA0oyQQauqBKvrUzjcxK91bbkI21gXNmYZC2Kq14WX9fgOlPldmRrMc1ahUt82Dis5RUWdb2uFzJsGTxWe90d1bQTcTFeRfsbznir3m7yGB5q1KMeuv9eZnKLRctNTuLY/eJH1qxJ4hnVgdxI+tehSxkoqyOV0Yt3RctfFjtGAScj1NP/4S1tufm/rXYsftbcx+qXHJ4ufqST9a0rPxQHIy/v1rbDZlGbtLQyrYVx1Rr2+qxzrnfknmpGuR616qdzj5bOxC8wqrNN70ndFWKc8/PWs+5uMdTWUmxxRk3t4RnmsO9vPmOTXNUkuW6OunHUoJuuZ+Olb1hbLDHuavPnPojao9LFtZSTxViPpyaxlIykuhMq9yanhIWqWquyXqTBie9SLkcmnfoTYc0gFQyT4HWrk1awralCecsetQLuZua4azTaRvFWRctxx61cjO2tI+6iGXLYbmrUt1OM1pTl3MpFqMnNW4Vx1rTUTLUQyauwjFAn5FqMZ6Vagj796VtbkW0LSwml8jcMmm11C5IkKkYqUQ8ZxiqitLshsTG00NJge9DKsVp5Ce+Oag3c5JNZN+8XbQkj5GaJGODmtNlqQU5Tk8kVE/HJosnqNorTnPNZ1xKASamRSKclxz1NRtNkZ9Kzvc05bEBl3PVmFN/XNJ3voTrYvRwgL0qO4jAWtnpEm92UnBzSKnPSsWtLlWHAYpVQseKdwZKITTXjNapK1ySjdLnNUJE+brXJU3NYvuMZKZ5JJrGS1Vhly0t8sPetu0iAUYrow6tqjOfYvxqBzVhJBiuwyauNkmxxmovNwc1nNoaWhPDPnn/Jq9DMo6t2q6T7EtEgnUmlZlbPStoyfUiSsV54Q4qjcWW4YKgj0rqpT1EYGueGIr2F1ZAQ3bGa8p8WeAbiyLy2a5QEkoetaVqSqK5pB/ZZw1xbNBIyuCGHUEVTkjw2RXCk1IOowTMmc0rXQPWtXqUixBcgjrRKEfnNTJFWa2IGX06etIY94yTSTsTLcryRgEjNRkgda0V2rGbKs90FBxVGSZ3bviuinBbh1BYmbmpY4juGRWl7aE2NazjwlWI1XeM1hOVilGyFuVCtxTFfjrUa2CC5bHoXw2vJJUeAdGQgA03xfbSC7LYJ3AdOwrxuaKxFl3PYWsTnUBibJqK+1by12svavbprmVmcOJjcsfD64EviSIAkbjz6V6P428OmC2SQoevXFeHinGGISO+im6d0edX9oUJ561z2sLiB8gnANd8XzbHn1r8xoT2bWIwBmsy6vH5HIr0bxSJgk0ZlyzMdxOTVCaV1NYqzepsQmds896YX5rVRRLXYOvfmkK8Zp2toK/QcqkCpIraSY/KPxp2tqaLRXNK20sIMv1qxJdRWgwMcVSguphUnfRGTe6o8zEKcCqBkLdTVSaWiMLhk9c0u/is+o0hC2O9OyccmiSuax0Qu73pfxqLdCn2DJBo5zSVkK3YFDMeK6fwv4RuNXmXKtt4NJl20uz3DwX4DjtEQtHzxXp+kaOlunCgYruoU7HNUm2zcghC446VYXA6muq3Q53cd5mBwab5uO/NVFEsPMOck0pkodhjDJzzSM/vTk7DInlxVeS5xk54FDFZFOe+A5J/WqE+ogZO8evWs3oaWZTk1QZ4YfXNU59ZUch81hOry6stQKkmq553DH1qncaztB+cAe9J1VbQtRMu817AJDj6Zqi+stIv+swR6VyV6x0U6PUzb3UJHJ+Y1jXt457nNeHVr8zZ6lGmlYzzO5bBNWbaQlvSuXmutOh2uC3EvywXqTWcqNuz071UZXjqYRdjXsbsuNpq38p5BrOT6kyVmQXEW5SQelZcssisVrSk0hRfRktkrbwea03f5cGuetq7Mu1yE23mcio5rHcvGKG+yIlIn0608hwSK6CK4UR0003qRK8jOv5CGyveoY5WK803GNtBRWgpbK9aaIPNNKpqrtFJFa5tWiOV5qAux4NOGxSdyeJS/0p7RKoyaHFNtil5FcfeOKuRtlPU1c2raFWJY3IFSHDrWF7vlYSWhm3oEZJzxVBnDnqM1b/AJSqWqF8sgZzmnQxkHOajnsmdCehZ+07ByelN8/ecZoXclQ6li3uGj6tV2K9YnrTp1NP6/zJnBMtq5dck81IiFx2pxehyNK5XurMsckVHb22xuRXNb3vUtT0sjQHQHHSpIJ9rYJ610yinZmV+hZOGGc5rN1C3DAmotZivZmIbZhL7VoRqBFxUOF2buehm3shZ9uagWyLDJ71rflWhXNZXRLCnk9DVqO42kVnNO9yk7otLcFl4/Go5JSa05knYzsLEG65qTByMmlKrra5SRMsYKdaqXVw9vyGI/Gik9fe2J0bsyzpnieWJgpY10Fr4p343N19a93CY6yUZnFWwmt0akGqpcr8rA0ST+pr2IyUldHnyi07FG4n5PNZd3c4z81ZzlbU0gjEvLs4PNY88xkfAauKrNLY66aNLSbUjDNWwx4AFcEmr3QSetyWBMcmp0qYq2rM7kqZzirMa+tOWuwr2RKOOaRpCOM1pbuQiJ5sCqs0xPespStqWiFSXbk1YjiOc1glzal3LMSHFXIoS2M1rybJGcjQtoAAKvQrx1reMbbmbLKAA9atRtgZq7pEliJxn71W4pM0NiZbgcVfgYDmpb1sLoXI2AGc05nHPNKW12SLEwBqcOMdaqEtCWiKTGeTULnPfFSxoYybqRYAcURjcpslEQUVVugVz1rVwsrCTM2WUKSSapzanGjYLD8TWd7IqxXa9EoOGBHrmqU8hZqiUk0VFFaTd6URxl+prCK6FXJY7Lad3WrcUOK1hFrVkPYm8wRjrVaabeetXUl0CKIgmTnFIU71ny3LQLGS1WYrcnmnFaibLAg44qC4j2gnvWnS1ib9DKuFz1qjKvJrllu7lx3GLyamSLPJHWoUNSmy5bRbDnFaUDbO9dVOy9TKTuWBLxyaabjaCM1bl2FEie69TTPtGep4rCcr7F2J4p8d6sJdH1z9aKWgmiRbpu5zTxeEDr+FbqSvdkNIX7b69amhkEmMnOa6ITvqS0TS2yPGcDk1z2qaXHPuDDPP4V3QqrqQec+Mfh1FeBpoPlkHOQOteW6ppNzplw0c6MpHcjg1lVp68yNLt+pnyxgiqrx4JrC9roVgETjkGgyuhwa0i1LQanYdFcf3jT5LpQvBqZRu9CupRuL4DvVCa9Z+ldFOmlqyXa5XLOwxnOakigbj1rZNJaGbtc0LfT3ADN0p8sXlngZrCU05WGrEttc7QckUougXyDSa5ikmTM4ZcnrVR5Sp61MdNBnc/DDUVi1GJWbvg/jXWeL4Q4WRf7pHFeNOHLij1oNOFjiJEKyEtUNzp8d4jbjg9q9qntY4666sh8JCTSPEsDOMLuxX0Z4ptUv/AAtHOuMmIHd+FfP5lU5cVZdjtwbUoaHh2sR+WzAt0PQVy+sbfs0nrtNenSu43sceJ30Ownso7gEFQaxNR8NBwSi5xXpzpqS13OKM3HRnO3ugzxE4Dce1Zc1k6AhkP5VlUi7a/wBfgdEZ30KjWgyaj+xtkkZNSpcu5SXUeLJs5qaKwdzwhNNTvoPrdl2DRSeZDgdasv8AZ7NOwxWsKTZnUqaaGbd6wASFIxVGW68/k9620tY52nYrsmSe9RlSD1rGV2wihScd6OD1qeVp3KS1FyM0oApa2NUhRg0o60myuoGnKjSHAGadtECidZ4P8HTarcKZIzsyO1e6eDfBMdnEgEfPHatqcG2KpLQ9I0zS1gTAAArYghCDFehFHDNtvQmOE/GmFxkmq8ybDWlGCc1EZxyKGykC3HuKk8/I680X1E0MaUetRtOB3pt3C1mQSXH+1iqN1eDafmzUSdirGLfajsz82KxbvWMZ+YH8axqTaNoJt2M6XWgM5esm919kzhgfxrysTVcla52QpvqZzeKwG2bmzVe41xnOd7Y9M1n9Z/d+ZrGg09Sg+qvI/JqxBctIOtctatd26/15nQoWJvLZ+vSqt5agKWryqt3L3TWEuiMpl5J7VJHJsxmtaa7nU3pYbdzl0ptuNy81c0uhjZkhLJyOgqeK8IA3VnZWK8ixFL5h609rASclaynLsZyjZjhaiHoMYprKTWcm2hxZJCpFPfCiqveIpLUZHNluKsqZAODVK0WQ9NGNYFuWqnc3QhyOlaJKV7EpXZDDf72wTxWnAwZBj0rGq3y2N1HUa/z5yKo3UOxhgVFKWgnGzLFoPl5FJckH2rdOyuTb3itswc1LFJt61MpdSr30LqhXiyMdKqPdGFyKmEbPUXdMrXCm4Gapy2TryBWjQ4z5dCSBTjDA1PsCLmsd2dDehSmcu/FSRKw5Oaptq6Li9LFmM5NW4AQcmslHXcVR6GjCwIyauWxBqlfm3OCRNKikVSYgNmtOVGUbseZvlqtJNtbI61b0RcFqWIL7AwaldxKvrWMpqWpcodSnJCu6opCVXArSMbLmI9TOliLSbiasRkYANXyLZl82gk8WfmFUZHZHAJ71nUg07MqD7mjakyJnOakwCcVhUkaJEmNopuSTWblqjRWLUAJHNV761L5OK6YRdrnLKXvGZ5XkyZ6VIbkqeDQpyTubws0aWnaq0Tjk/ga349QV1HzAnHrXv5fibx5GebiqLUrogubjIODk1jX11gnBrunK6Maa1MO7uiWOKNPt/OkDMM1503fRHUvdR0MSiFB0p6PvbNYOyMPMtRnIqVRz1p9CCeFQTk1aQYGQaIN9RPzEkfBqNnqtQ6WK08lVdxduCcVyznbRGsVpqWYI+RV1AMdqqHQiRNHjNX4cAD1raNr2IltYsrMFFOFzjoauUlEEiWO5OOTViKV3YVLqX1QNJbl23Dnk9atx7+COad23Zk6WLcDkHmrsE+AM1XQhq+qLiXOcHNPNxnvWUpdBW1HJKe3epRMVHWhMVughnJpA+eprQLB5u0U0XIz6e9CbTDluh32oAcnNVbmfdW3NdWYuV7syLxiQR2H61zeqGRSzKehrgxErLQ1RmQapIsgBY1rwsZVznOacJ3VhRJxEDUscIXtVxWtwbJxgdqQtg5zW6ROpHIxPQ1CeWyaxnqzSxKgGM04qMUaiaHxQZIIFXorfA7Z9q0pLuRIlMQAyaz7xuvFXJWQlqZc8ZbJqjPHiuGcdNDWIyJOelWo17mtKdrag9yxFwM1Mr471rsQ0I1ztqCW75681hUqN7MtRIGuDuzmpY5d3U1lz9i2iwkmO9SiY461re2pDbeg9Z8d6d52O9VzuwmrCiXvViGfaAc963pTRLL0d820VUuAZjk+tdU5qysYxRWntFnjKsM1yXirwRb6nAwdMnH3scitKda65WUtzyDxF4Pu9Inb5SY/4Wrnmt3UnKmsupVrajRkU1mTnOKEkybJlG4cKSQeKqyT57110o2Q7pFSQ7z1oCZ681s49zO+patbbJyQanOyJunSoDdlmO83LioZG5Jz1rKS5dGU2Qk7T3p0RJOaVmJsmMhHqaikO44p8utxp23N3wbdmDUkOcYNerauv2iz3t6Z/SvIxcl7W/U9Og/cOK1CILIeMVmzyNGpIbpXdRk09SasVy6mYNRkivopcZKsDzX0X4c1iPVPBaRswJVSv1GK8jNY3qRka4GSUWjyTxEmLqYf7Rrh9XkKBxnOQa9PA3srnLitzvYy3Wp1CuOa9dq+iOAgnsopDyM/Wsy+0CGZCcAH2HNRyptXRUZNHI6pojwSnCkCm6Vpv2mcIw4z+dc9eny3R20ZKWjOiTwigj34GMZqndWtvp4OccetY4eDqS2/r7ia00nZGHqWtpGCsZArn7q+kmPLcV6fNyKyOV+ZWOXp+MCsJtgwDkdaN2aSkupVuwFcmkCHNF7lJCmMgcij5h0ouil5iqWHJxS9ahxV9BpD40ZztHeux8G+DJ9SnVnQ7T7VDutjRaLU928G+C0sokxGBtH516Hp+mLDEAAOOc969ClHqctSeprRwKFBxgVMpCjNdKvaxyMjkbniq8kuDmqZe5C0vrUMk2Oam4PsRG4I704Xm0cn8jRFplWXQGvCVzkVVlvwoJ3D86TdhJFF9XTJBY4qjeaoMHBrP2nNoaOPc53VNTGCdxzXL3usHJy3P1rjxE7I6qEDMuNVmIzWZPeyzEjJ5rxKs7yuj0YJWIVhZzuzUwcbdrVm22rMG3Ijcd+1T2c5VutZ7ami1RrxXIZcdKiuZMqckVzu1xctmYV1IyyEDpToD5jAGqUt0dkdrk1zbYQMOaro5HTNSm2iE77kgLfxdKkji8w8VVtLsHpsWLVDHNjrWuGUKCawdJ3uTPW1iNpQ5xjikaPIz0FDhZGWzI9+3iiT5k60Rh7o2itETG/Iq/FOCOa15ULluOYqR9azr61MvK1mpWdh2aZRS2dGya2bF8IA1Z1Z8zN1qiZiN2RTJo1cZIzWMVyslphDHgYontMrurou+UyvZlNoivGKRYTnJNQ3qX5onjcqMZqOaDzDnFHM76g+40ssY5FBnjZeSK7KULq7MnB2uQNIm7ims+8cUnSjTdzS7SsLa2G99xFXZLUInSueVpMrn1GRW245qykBA4rJPUqctCxCjDrzV2EEDNdLUTjkx0rjb1qi8mG5o8xQWow3HvUcjZPWpqJNaGyi0wTI5Bq7ADt5Ncqi1OxpP4SK4O1qil5Su5K6OdvYqgbjjFPWFs0Npbj6k0sR8vgVlXtu3UClJtvUcZWaHWFy6fIR0qd5zuBArlq9zrhYmS5Dr1qWFgTyaFG8dCZuxcilQDk0SSoQeK7lG0bHHJO5k3iEvnPFVlQEnJrCfuvQ3hLQmVvLGRViO+kXBzXRQr+ylzBOKmtSyt8ZI/mNZt/NwTmvalWUo83c4/Z8srGTuMkmCa3tMQIgJFcfcqorItSzZOM1JbgkZrHqZbKxcibPJqdWzVaEWLMK96lzitL21IbEJ79ahlbAzUy7hEpyuSaIkxya4nZyNeliyDjmp43JHWtnJpKxNiaNyOpq0kpxzWqa+YrEqyFu9WYIGlI4qJ3egbGnb6fnG4Vo2+nhADjmtoU7GMpGlb2aZ6daux2QK/T9a6XFcrMlJiGzx2qNoyhx0rG1tCo3FDsBjNSCVgBWL3K6kiXO3rzUguB69abfYHEr3mppACc5xUNtrDSNjrk9M04SSdjOSZoiQyJk1XlcoauS10LTuhgmIHWmSTEjJNOO+oFK5fdWbdQiRSO9c9SF9UCujGk0kGbcR1rRgj8sAc1EYWZXUtJgCnM+O9dDS5SUtRvmZPWjORTUupQEgDmm8Gs3vcNSRelPVM81VmBagWrsQArSF0ZsWY/LWVdnJNVUaYIz5qpzKfSuaa6I0RCow3NWEJNKEdCmTDKimNLirempJDLcZHWqryHPJriq90aR8xolNWI5fl61mlYpkyz1Ksue9dPNsRYcJeKcJTnk1Lk72CyJBL6mpY5TnrWlObTsyXAtRSk9+KsIQwzmuyMrrUxkrEgi3dKbNZ71wRxT1WqFe25h6l4Sh1BTuQNn2rmtT+FFtIrP5IJJ6ih6vfU0T0PNfGXgKfRWaSFD5a8nArgr2UxsQRXTRi72Yp6bGbLMznFRiJmPrmvRSSM2+xL9mOM7alS2HHQVjOQrXLKsEGKqzyZasrlKILJsHNAkLt1q0ubWQ+XUJAy89qRZNvehK7E2noTrcgrzTXBI3ZonGxK3LuglvtyAdzivborOaXSIpCMhoxzXh45KNRNbnrYbWByuoWWWbHPPFYl9bbOMGtaUnoTVu0zmtQ/duT15r2P4P3Emo6PJblslV3AVnmMFKmpPTUjB3uzK8Y6Yba/mUrjdzXnuuWeQ5HoaeAneOgYtaM62x3z2YmBzkdM1Tn1Z7dwpUg969rDYqMpNMwnQ5dCzBeTXK5VCfbHNWDFcgZMTdM9OlayqRvYj2ehl6hDv/ANYoz61Qga1sG3sQCKHTVR3J53T2ZX1jxqkURSMgZHrXGan4glvXPJq4pU9EZtuTuzLaQuSTSGs5NvUT3FDbaUscVEncu2gh55NC5xS5e41dDgcnnpS78cClfoaJaBvJ607dxzRIAB7mnRjzHwATRbsO19DsvB3hCXUZxI8LFeOMV7t4O8ILZxp+7GeO1a06d9WRUdj0LTtPWFRgYP0rWjQL0/Wu9RRySeupKTgYNMdsD0FWyCGSTiq0kuQelQ/MqJVeTAzmq8tzzk5obHa5UlvQOpHtUB1D3P51DsXYibUWAPzcVVub8MvLYHpmiWq0GrXMK+uZPM3ByOexqu9+5Q5PNeNKs6dSx1qCcTEvLtixDH8aw9RYHJ71niKq5dDohG1rENs6yLhqjuQin0rzna9ze2pWW55xUiW7SnOetVLui2rExtSq9M4FRwRESZINZSehUGXS+xeKryT7iQSaxTs7mi1IPs+9iSM+9SLahOcVnNu5cZO5YjYMu0rVe4gVWzximotO4m/eI+q5qS3f5sc1o7NWKa0LsSgnOOatMjFM5qHfcyluRxKd+GqxMv7s4pRlzEVNzN89g5DCpfPGKuUbLQp+RGzg8inxErWaTtqNE5JxSxDcfm71DbBDbiJBnA5qGNytc9RuLubU07akiyljVmL5gAad+fcqSJgAvamyXAwQa2prSzOdxuyqV3npQ0eKmLtLQb0EVOeamMQ29auavqh3Kd0meOtY92skTHBOKKNWSloWtNCn9omL4OTV2ydyw3Z5rtnO8bCaVzoLPAQGrEqB14rjcLaGT0loNhhIqysfrXPZ8xUnoSxxgipM7Otd0Ujke5XupNozmsu5uSmSc1M43TNae5WjvMv8xqyswkGc81zXlbQ7nDqKJCDyKtRXZHWlGXvamc43RIcSc9ailOAa3U9Tma0K8QzLVqTEYzSldguwqSiQYNNntQ6GrnumiW+UoRwiNzmm3Lqq5JApKipas0UnfQpLdFm68VaguWY+1L4di5asmNxIozzSx3EjnnNae0vHcXKrDpPm61F5WDmspu+xC0ZJ5JYU4RjGKTTZd9BDle1VLtS6ng11U69o2Ia1uUrWLM1bMcojQCuuMr6mNRO41ZCz1ftn6dhUrW5m0W0bJqxEeaTWqRmy4h+WnBs1s+xldinBFV7jpSlF8o4lRl+bPNOziuNJs1HoxJ5q1HwM1ond2FJWRKrc1YgBc1Db5tBGla2+eorVtoNhGAK6oxtqzJvU0bdenFX4RW7aM2W4Tj0q5E+apSJsSdaRoN/UVE7NWBDDZnGcCoXhKnmspxtqh3IiCDig7gOKh6Dv0Kktibh9xzjpVi101IMHIz6UoR1uJl5DtXGcYqGZxnNby1BLQpz3KxAk8Vkz6+gk2bh6UcyW4mnbQnjuluFyCKjk4NJpFogbBNJ05zWNtWVYaZSO9JvJFVe4WJE4HNOLUdAGbixqWOLIpK9wZOsRNTQQEmrabE3oXI4CM5GKkWMrWsVYxI5s5NZl3361M0y4ooyKTzVeXgc1iy0VifmqxFwMmlTu5WGx0kgA4qo8hJqqkrOwRWlyJ34qFnrlmrqxcUIGxT0k96xincvoSpJk1MsnHWrT0IsO8z9Kesh70Xs9QsO8zvU8Tk9e9Ln1uN7F2FquQnNdtOZhJF+Bc4zVo229M4rriuZGT0dyIwFDyKftQryBnFY8tpFPUwPEPhuLUYHDRqVbgivnr4n/AA8l0md7q1jYoTyBXVCa0Y91Y87Fowfay1ZitwnOK7ZT7GJM6YXpUBUg5rmjJ3uasQkAcioJQM5xV67i1GZA4600Eq1Wu4tbku/eORmk2Y5Io2YpChMt7CpcnGADTk77grF7QZBHqkO/gbhX0boKJe+EYPlG5Q8ZPf2/nXz+bKSlFo9fAu0dTiNVh2TMMAYJzxXO34BLZp0G5JMqtFWZzWowLyT1r0P4A3qDWBbHkO21snsc08yTVBmODf7yx1/xY0I2s4nKnbjrivHdahGHIU96wy2pyppHTi4NlzQVv44QhHye5rXj0+0Zy90wyfXivosPhI81zz8RiLOyJX1jStKGd6cccGsnU/iVp0SskYDE++a0q0Fz3MKc3I47WPG73LN5fy/jXPXGuXEpY7uveumMlyitb4ii80j8s2aZtJ5rOUlKRHW4oQnrS+Xxms76BYNhNKUOKjQvQAuOoFB454onZK5aSYZ9aQHJpR3H5IAaUN9aT3GthUDOcV13g3wlLqVxHIynGelXazsgslqe8eDPCKWsSZj56V6NplgIQPl4ArvpQVtTmqS1NWJNoqUnjk1tZI52yJpCPaomn4xmmNK+pA83HWqss2Dmp6lcpUmmIFULi5wDzSuNaGdc3nJOaz5dRIJ5xWDubKPcryakSeDVS41Qh8bqTktjVQImmMoznNV2b5uTXiYuS52bx0Vivc26yjNYep2MgfnGK5JNOOhrSlZ6lSKDbz3qvfAhc1il7tmbdTODHPWtC0u9mM/rV7LU3snoaMNykwwaSQKmSK5JJO9gULMhEu84PWmPasTuzim2uVFNKLHBQq80+JwTg8VU43RS7FiONSc+tRXdpkbhWMb2vcV7MpLDIDipo0MZG4Zqm1ujVtFoOqjNOF8MYJqoS01JUWxi3Pz57VajukdMBuaiGt0RUhoVJ0AbIINV85JA5rWPw6GUb7D44mJzV6C3GOalu+45MdMuykiOetZSSirscdRJk3LUCRkVyzd1ZnXC1rEqxkjirESEUl7rvcbWhLg461WmRt+BW/Npcx0uSx25C55pxhPcU4RclqYtjDHg0AHGDTindgpFeeIhs4qpPb7wc01Gy1RXNoQRacC2SKsizRDkDmhxa2/r8BczuW4VO2pULb8E8USk7XFpcuxJlRSk7c0oxTJaHRzbTzTbmcdq64RTVzJrUozz7u9UZ1Mmeayna1jWCsVGgdCTg0Qysj4Y8VjCOp6EZJxNNSpXrUcknlnNZOFpXMd2JFqQztzyTVonzEzmumC5tWc9WNmRRcOKuNGZB1qtbGcnbUWK3K1My/Ie9JL3jOUkzLuVIY4rLvNxPNax00NYsjtotzZrShhwaxqPUtvUtiFWXnGaURKi5rLlbv2E2yGYjrUDzY6VaaVkNK5YtpNw5NWPLBFRGom7IqUbEbx+1V50+U806bTZDM85ibOKebsbeTXdCfu8qInG+xNaz5bJq9Fce9ap2MJRsy7AxOCTV2EkEGqurmUizvNSxZJzWkbNmctEPYVWncitKvw6hHUqk5NOUZFcK12NR68VOhOKhvlV0DJ4ULHmtWyt+M1VL3nciRr28W05NXEWuy5m9y5bRsSOa0Iozj601qrohssxRmrUcZp2d7ktolHrUqDPJ4peTEycDIqCaDJzVy2sSQNbDJ9+lRm355rJrW7LTQhhNRsCDxQ7jT1GiUjqeajlc5zTsDRl6judDXKXGnTtclhnrXLWi56IqOxsWW+KMA1OZCT1raM/dVxpaCc0mCRS9AEERPWjbtp2sF7jlbtQQSabdkFh0aYPSrUa9qIK71FIuRJxViOME9K0S1uZsspGcipNm0Z9fWtYK5OlyrckJzmsi6GSaiexcN9SoR61UnI5xWMnoWtWVW4OTSi4IrOm9bsbVxjy7u9RGTmier1K2I5JOKhLe9Z8tilsJu9acHrBrsUSI+KlD96uKCxIslO3ZFJrW7BIejHuatQnkVjHWQmX4m6CrluMkZ711rTYxsa9jHn+lbEFtnsK76EujOae5FeWxHJB571SKDd3pV42loWmKYfMGCAa57xH4Vh1KJleMNuHpSho7At7Hinjb4UvaSSTWaf7RA6H6CvNbywmtXKyIVINbQlZlO25BggcioJJMV0NXWhKdyLczUx1PejRajbSQxhjg9ahbrgVpTuJkqE45NSNuC4zxTlZag1qRb2VuKsRtuGe9KUrLQWlie2fyplc9jmvon4ZajFdaBJCz7sBZB+PX+leHnCvGLTPTwPkYPiSMx3sqHHU/wA65q9t9yk1zYaV4Js6KkbHKayPLJNanwn1o6b4mjYnGWGP1rtx0b4ST7HHh21WsfRXxNsRqehLcoAwKBx+IrwHX7YIkgxjr1ryMuktmejiVdWJtX8SWumoUjZMqOOa4TWfGE88pKSHnsDivvfdWx83u+ZnPXWrXFyfnkZsepqm0xJJZqhtrSJd7aIYZQ1IcnuMVk13DyFCmnbWBqdNxJ6ituPWjDAetZKS2NFYeoyM0vSiTWpfLYRjmmEGpv3KVrCc5owKrS+hWltAC804Lk9KUk7CSvodN4T8KzancI5QlSRXvHgrwclrHHheeO1aUY3d2TOVlY9I0vTRbxqMemK2IU2j2r0YrQ45Mm3DGTxUbScZzV2IIZJOfWqsjnHPFFxrRFeSU+tVLifHei+li0yhcXR5yfpWddXXB5yawZokZF7eAc55rGudRCseaynKxtGKKb6qMnnmqVzqZJyDXBVrOKvf+vvOqES3YaqSmG/Oi6vx2PFebVrKe+/9eZrGluQnVQo4bmql3qSyDGQTWMZqz/r9TSNK5USXdIORVyXTRPAWxk4rO6e4VFy6nPXVlJBMcr3oRSCN3Wib0sbU5J6ov2m0e5qxJhl5Ncz9TbW4kUAHzBqlPC8ms7vZkyd9ypM+OlEbAjJOBW0k2rjSsSrde/0qVZt69TSeisDQgIU5xTZZAR0qIr3tQaCNDIMik+ytnJFC0nYpytoRyIwqs1w8LdcVUVyyaHuO+1mTvUkUvzDOOtac3KjNx1NOFFKg8ZqUNjpXO7yfMLlBm3jmmrHjn1pV2hJWHlN3FKloSCawqK8dDSM7Eq2xUUqKVbkVkqb5tSudMsqilecU1oFHzGtoRVjGchyshGOBUEx2muyKsroyV72K7Nk9eKduA781k30RpYbIy49artHub2q5bEpdRVhwelOKZIzWkVpYTZMgGO1NYjPFS42Wok9S1BMQAKmYhhzWKT51cue1yNlwarupY9a6m7LlXUyixDaMwzSJZlTk1hy31Y1IfJYhhms+7sfKO5aIU9br+vwNI1raFLz3RsE1KXaRMk05tbM2T6mfNIY5cjitOxvN8eCc1pR2ZVWF43LSLlgc1bSTii7TOOW2pIJwFOTURu8jArVRV7mdipNIGOc1lXrZJxSkle6NIasZaA7h2rYi+51rKcbo2lYdvKiq8t0Vbk042SsStRPN8zoainVguRXLU0dmaw3sOsWbdzWxCuRzWUnbYuqlYJIcZJNULo7c1pBJPQwbuZN2+QeazJbgxnrXfTaQ0tCxZX+TgmtW3nDMOeK25rq6/r8TnmtTWtpfetCGTJxTjrqznki2nTOasRdK0W9zKQsjZGBVSc4qqsrxswjuVi3OaUPXHFo1RYiXdVhEzVtLcRoWkOcZrZtYcKDWlKKuZT3LqL04q1CoY1s01sSaVsp4Har8SjHStUkombvcsxRjOTzVhE4pdCbj9lPC1NveAeDgUEgDk9asViOSRRx1qu0gLVOrKina4bwB1qJ8Ghi1K8oAyahcGhF3K00QI7Zqm9su7JGai2oajWgGM03yueKhx7DT7D1iHrT/ACBVJW1AYYsCo2j3Ghq+o1IQptpVUk1LegyZE71PGpNNaMlluJDVqNMd61TdyJMmU4psswRea0vZEpGZdXGTVN+e9Z3uWU7hiucVTc7etZlLYqSyc1EZc1jJWdixpkwKYZc0Ss1Yq3Uids81GXJrOctFcpCg5704NWbfUdiRXqVW4qabdtQHqaeMk029GwJIyAeauwVmpailqjQgFaFtGSwwMmt4u2pk0bdjGTj1rctI8Lljz6V6GGT+ZzTQy/hymRnFZXlEOTninWb5rjitByxn1pssW4EHvWab3GZOq6NDeRncgyR1Iryvx38MEvA0sMe2UZwRxmtHLmjYuNkeR654XvNJ3GaIjHpXOyRFetbUKnMrdSJe60hqQnrUzQrt5rb+6xrVWKk8RHIqoQVbJrSEkrkvXclibJyalaRQMdaUtSrdCJgPWmxkqetQpWbYWuWhLwOa9d+EGpPKhgLZEiFAB+debmP8K53YJ2NvxGpF4S3df1rCmjEqnIGK86nG8bo7Z7nIeJICu75Rn0FYHh+6bTtdgmJICvzXqOHNQce6OGL5KyZ9c6bdx694Bj5APllD35rxLxdZmIT/AC8rk8+vNfM4ByU1zbnqVWrXPEb/AFqe6kO5yQTmqDyE8k1+jyPmtkMMnGetQSSZ9qTTuJSGKTuqyriiSuEdSxHhhTyoHNckm+axqoocIs0jp3rNu41uIVwKbyR1o7loQjAzSAZ6mlqN6iEDGaZkZrRMEx6knAxW94a8PSarcr8vy5x9anVOxR7v4H8GJawxtsAJx2r03SdNWFVwOld9KDSOWpK5sxxhRwM1NuH0rptoc9xrvgZJqFpe9VbQZDJKTyOKqzy+pqXpoOxSmlPUE1RuLnOcmpbKRmXM+M/NWXd3eMnNYyepqtdjDv77bn1rDvdQ6kVyVZanZTijMlvD/e/KmRTs0mTk152ImdkIrc0IpWC5HU1FNOzdSRXk3eti01sUZ5JAetQh25JPJoU3bQ3i1YsWylyDmtq2uGjTGTik6m9zOoubQhu9kvJWqEkCuPlHNZIiMXHQgSOSJ++KsneVzS1vqdTkrXCF26c1MS6g5qHK7F11KU8mDmoDc8+9dEXdWNIxu7kkblznNX7dc4JNZSkoPUma0HSuqnk1G0qsMGhp3TRCuS2koQ8mtJFSZeKUncmSa1Kl7asvIGayLlMg5HNawXMtdy4yvsR28JPJqz5eGBJpTv1Kv7xo20mFwae8mOTXNqpWYNajlbf0qwikgetKSctDLbQkSM56VbgjIPIqYQd7EzaLAgXGajkt1I4FbOld3MeZ3G+SVpsi5XrWd/etYfNcqyfuzk1FNNuH0rsim42GnrcqvOBUZn5znmsUrSNktBRKfXNTQ5JyaTTvcOhOxAHSoW4PHetW+pkKqsadsOeajm5mDdh24p0OKctyQOTVJLmBWaGvdZNPjYMcmqvd6EuLsWFkAAzTXmHrVOK0M7MT7TxVW4bdzQmk9CkjNkgMstWfsTeXWNSLZupcpmXtky5YVXtZWik2mpoycbp/1+R0qaaNaK5yoIp5uioq1U6HNOFkRNqHvzVZ79y3Fdad1/X+ZEY9yWJ3nHNAsHZ81zV6ltEXG0SVbPy+AKlAZFqYXvYlu40sWOM1BcQk8irk9fdEk0ySxhLkA1flsd0fSuavG7uac1iK3tBG+autIqJ16VjGEm9RybZUlvR3NU7qYOvWuqMLLUixmTKW5NZl8Mcit4rZMq5QS78qWtqwvN+Oa6ZKyRnJXN+ynBAya1rRt7ZoTZyzVmzUiHFSg4710RjoYNDHccmqsrnk1jVKiiuzH1pEfc2K5HboaJFyJ8AVbtgZHFXz2QraXN7T7bOM1qpFhRXfTirXOeWrJVU5q1Au3BxQ2K5ejmCjJq1BPu79aL9BcpdhkGPmzVtHUgEGtEjMcXAponCg7jgCjbViWojXa+9Vnvc96ly6l8pC9yW5zTRKT3qOZlWHmQ0hkxVeZIxpKid8daEPcqyOe9QlsHmpiy7DGYA8mo93zUpMQ/zADT/Mz3pwXu6iaGM4PWkLZpSZXQZjJp6Lip3BssKlTRqM96ZDLEZwKsKfl61qiWIXwM1Uupzg05AkZ0j5Oc0x5goqYsp6FC6uMk5OapSz9eazmy4lOWXHNQmasm7u7LsNMue9JvqL62KsRvLTA+aiouoDw2KcGJrO42Srkcmnq3NCkthJEqsQc1ICTzSkrrQaJoULNWnaxE8kUQpXd2TJmnbw7hzWpbQ4PTFdUYWMJM1LNSMVr2zcc9q66TSWhlN2JZxvj6fj6VmyRENkDj2qq7vqKnLoIEA+tMkTmsL6WKe5BJH6iqNzZJOuGXr3pp2KurXOQ8U+C4NQibfEG3cdK8T8XfDK806SSa3TcvXA61qnFPmQ37xxMsL27FZF2kcc9arySYHWumLu7i5raEB+fOTVeaPBNNL3idyDdt4NIJuea1YRsO8wU4etS73uO90LuIYc8V6R8JNSW31CEEjCuD19Tg/zrjx9JVKTR0YWpaR6T4rgDlZF6jIrmGUrnNeNRk1HQ9KUruxzniCFXLMeo9K4e9Itpwy8FTxXsYbWmcFV2mmfSHwW1z+0fDEkJkLYUOB146Vz3jy2Ecl0mOME4P418xCDjXtfqes5c1NNHzSYSfeo5Fxwa/QFJtnzDXUrs5zQEJOcVpZ21AQA5zgipUB70minoW4XHAqYDnrXLODubxAko1NaQ98VluNjTJkGmhs9TVcrSY0hcjbyaaTSSd7latahkYpBHmqW4orc1tE0WTUZ1CoxGeuK9u8BeDlt44yyA8DqKqEHKSbFPSJ6xo+ki3RcAD8K3II9gFejFdGckmiYtjvTC9aLQhbkbycdetQPLjvxSux2K0kx75qrNPSl7xS2KVxN1PNZtxPtUk5zWcmX5GTd3WcnPFY19fKASDzXNNvoawRg314cHJrEuZ/MbbzXFVnY7acRotZWGVGc1bs7F9/zjrXl1qt5WZ0OaSsbMNrGsfOCaz75UVsCsVT6mUG+YpSRbuKabRlXIFZNrY6LtaEcZaJ+eKsfb+2aXLoapKWo83QK5zUQuPnyTUxiUoltJFkXOBTiE244qeRa2JkrEa7Q2RTpZRtIyKJU+V6E31Mq4f5jzVdlOaOU6ok0RZetXYpSFzmiVnaQ5JWIppSDnJpiyE1V7oFaw4XOxupq5Y6rhsE07aXZDhdGp9sSZPmxzWbfxqeRis1G2zMYpxZWU7aRpeafNc2itS7A37vJPJpGmw2DWdTW9hauRPA3cVcWUDnNFPa3UxnuSpcY71Zt7xW4NVCNpXZLjdF0SArQx4zW/Knojnk7FK4u/LJzUIvQ3JNQqavqWldXIprlHHWqU0hINVZxd0Ul3Kj78896i+cNnms76XNk9CeGXdxV+3I280mncmV7WJJGwKjXnmn0syCaMgU44zxxUXs9AsNlXjNV261omrDSI2OW61btW45qHvoD2J5VJ78VXwc05OWyISEwc8mmvmhtLUBiKBJmrvy+X2qoe+wnsZ17s2kGsVk/ekiqlTV2zWm3YtW4IXk02Z8A1DjZ6FXuUVlZ5dvrVyO1LCtpu0RSVmadjbbcZq6YgvNc+9uYylIryyBT0FVpp+K2jbdDtsRW8hkkyKvCMSLjFc1Sb6GriSwQiIirAuB0rFS59GNRuRy8HNV5CT3pydmkhWIpYQVz3rPnBQ8niuindojqU5JRnk1SuR5gOK1lzJ+QzIvbcq24cGp9MuCmBmuqD5tGQ1fQ6TT7vdgZrptMPAJNCWu5zzVjVRwBSs/NdTZzW1GM1QyZINc01fVlpFdsnikXg5rmfkWkWIjmtbTIyWBojrIJ7HSWMZwCe/WtFUJr1Y/DY5G7EiJ3zUoYLUPRBuKJMmrEUxXrURa6j1Lcd3g+tTC+wOorRytqTygdQ5zmo3vzk4aodS6Glrcgkv2JILE0z7ST3qL9i2hyykng1KstVe7RBIJge9NeYY61cuyEkMMx9ajklzzSTGVnk75qJpfU1KKI3kzQpLGo3G1bclVcYJoYEfjWlmS7DSDTlQmsmmJ3HiPNSLGc81UfMomSLNTxw4rWKvoQ3rcnSMDnFPIAB5qoxsRdEMkmB2rPupjznFTU0RUTPkm6kmqk1z75NQpWRaXUozTc9aqyz+9Zzd9C0VJpwe9QGc1hKVnqaIBL3pTNmlzis3qNZiT1oUnOTU69StiUEkc09Bjk07LcT1JQc05Tg8molF81wSsSIeaswjJFPW9mDRoWsOWGeK07eMcd63jDTQzk9TSt0xir0B561pYyepoW7gYPFXIboAZyT7VpCSTsZSTZZWcOME8U11+XNbVNrCWhXIwelIRXMa9CKSMmqzRE8+vSpcugX0sMe1LqcgH61k6t4ajvIm3IMn2rWmSvI8f+IHwrWXfc20ZWQAngferxvVtLmsZmjlGGU4IxXRCXK7DfvK5mMSD9KGwV+9zW6bVhXKsnU1A4rVQd7iTuxVNThsihlRtqSJHv5Jrf8G3v2TUOGwRyPrXPjI3pSNsOrTPcrpxf6ckoOdyK3v6VztxGqM3P4V89SXLGx6bd2c/rib4mArz7W4Ssh5716mElrY48TZHqv7PWrGO7FozkLIGTn36V1/xIsWQNLj7ynJry8TFQxLSZ3wd6Vz5dKENTGtWk6V9hzK5887XIG01s5J5qIwmM1qqilpYm93YBmjZls5pJpMtrUerFKsRSe9S9S4Mc4Lc5qMr71yt2ukjbcaQe1MbIqrtsL9BNx6ZoyTVdQvrqOTca0NK06XUJ1ij70pNIcWj2PwB4LWJInkTkcnivX9E0dbdQuK6cPFrVHPVlY6WCEIoIGMDmpxXarM5d9RjN2qB5MEihlpkEkpHeq8kxwck4qW0NMrSzEZ5/Oqk0+B1qXIpmbcXJyTnisq9vQM5OazmXFamHfX/JAJ4rFurwtnLVy1J2R10omPeXJboaqQjdLljXn1Zq51wVtzftTH5YzjNNluETODXlVWudtiadyB9SwvDfrVC4umkbJatHNJG0IWdyW3mVupq6kikVyyinIuUWUb0AZIHWs4ud2O1U3Y1gtCRpCq/eqHzmY4zRBu+pUX3Jkunj4ycfWnNfuT14rSErPUbSY6O9Zz1qR5mI6iobTJlFIqyBmOQaIgc/Nms5XsaxloSFCTkVNGSoyxqYvZMtvQimfJ4qMO1W7R0JRFIzHqaktyfWmpJj6GjBKyjGaWWcHgmoe9jNrW5Du3N1pwQnmi2mo72LEL4HJApJADzmsaid7olPXQILnyzgnirqSh+Qa2t1FVj1Jlfnk1NGw3ZBpPXVHO2W0n2j71DXuAcmuumk3dmE1czb64L5Oaofan9aiatLQ1prQb9pk3dali3ydawqVLM2styx9nxjvT47ZScmsObm0JkxTYgHcKfEm3g9a0im9DNybFlGVqAHHU1pDuxxeliWKTJ61NnJ61DhzPQbdh7YI61WuBgcVNWLS0EmVVDFutWo2IHFKDvuaS7lqOfeMdaRgKtxvqYtakL8HOcUn3zyaUlZ2NESx2+eRTJVZOM4p06nLqiJGbdq5JyTVMWx61q59ik7EiZTjNQ3HcA9aE2ykQWlpI02a6C1tQqjNRKd9CqlkWPlToaY8wPes3oZFV2Lk4qvNGzcVXtFEpKzH2cRQnIq8p2kGuS/vG0iZDuqRIc81FVX2JvYSRVPFVpk2miV27lIifO2su+fr2rrpysrEdTPxv7802WIqucirlN3sXLQz7gK2Qap7hA3WuiDaZnJGnot60kygHHNdzppJQZNdFrO5yVlY00JPepDnpmnzM5xQOKjkXg55pSuFyuwx1pgHzda5JSsaxLtqm8+tb2mQEkdOO9a0YJ7bmc2dDax4A7mrgXA5r09EkczELVG8nPJrCb0GkIs3PWplnz3rNdirMkWY54NSozuetVdjJhGTzSNGRUtMm5GUJbilCYpxQ2x+Md6NxzmrEhwc5ppY5PNJ+YmxA57mmu+R1ptdRtFd2JNQPuJrLW+hQsSEtzV6ONcVpCOhMhH60w5zVCS0ALUqrgVNrj6WJFWnBaLdSG2mTICO9TKcDJNbRshMHl21C9zt5zS5hIpT3gOec49KoT3GcnJqZSvozSJRnuOTVGa5561jJ2LSKVxd5qnJdE8A1lKaSsaxjoRNIW703f71zX5mxtBvYUqbjzQlqN6Ey46k08AGtd0Tcljj3VMIeKpQdibhs2ikznrUzl0KXckiPNXIMmpha2omaloQcZ61pwEAiuynojGRcilHY1ZSdR1ah6kk6XY6A9Kkju/m4apTs9A6F6O754Oc1ZjnBGc5zW924GajqKcHJPGaYxzxWOxQhG7rQI+/wClQ9XoDAR5NL5IK81a8iWZOsaPHcRN8oJIPGK+evjJ4Q+yO15EhXA5wPSum946jjdXPH7hSGOG/Go1UsDk12WXKmkJuyGPET061XlRkOO9HO3oguhgUnvShSp9qcd9RX0LNuCx61fsWNteRuTgZ5+lTV106GlOXvI9s8MaiLnQouQSgKnmqd8+WPXI9q+c5Xdo9i+iMG/O8NmuK8Q2+GOP/wBVehhkoysc2I+EufDTxGNB1uNmbADA5/Gva/FGuwaxpLyKRkr9eormxlFe35rGuHq80OVnzYLIoMkUqw5Ne+5XdzyEtA8lcHNZ15blckc1cKlpWsQ1roZ5B3c0/djviuiTT2G29hM570K3zcmp5tbApWLMcgI9aV+eaylpqdEXpcYeaYeuDzUOSEJtXr3pVXNJSd9Qbe7J7e3M8gRRyTXqvw88FsxR5EweDnvmk37yixrRXZ7Z4d0VYIlAHT1rqrW2CKK9GmrK5yTd9C4EKjP9aazYGa3TXQzIpJO+c5qrNJ70DSK0kvHJ5qrLL15rN7lIpzT5JrOurnGeaTNLGTe3xHTg1hX+o4JyawqT6G1NIxrq93ZJOKzbifJzkGvOr1kjrpqxQmcu3vT7cgcmuDmu7Nm5NLdeUud1Z93qDckMawn3LjruV478ucGrSybhkmpnK6sbqJYgyTyauKSB96sueLdhvcZcfOOc1Rkt2+8M1DemgRdiF0b+KmqCD1rVSTNNBJJwpxikV8nr1qpW6lJE0fBqU8rWCl0ExqSbWweauRwrIu4GrlHQT93UZt2tg1JsyhrLRKxbfUqOCDSoM9TinbmQ2Pe2DjIqtzC/NTGfK3ccexYSbcOtMlmPrVcxVh1rIGar3mIq803LXUyndkEk6jkULcblxVP+UmC7jGmweKmtrsK3JpXZo1eNi6t1u5BqxFcA96UdHdnHNE6TA9TUc0nPWm5uLuiWVpcMMk5qExmqnMEtA8nHarVsoC5rkqvVmyehOJADg0/tkVdPuzKTsx8MhY4PSpvKHXFbWV9CG7EUuAKqTDcMiqtaPmC7kKTNG+DU4nPUGpU1FX6mko6D1nLDnmh23ClJqSJsMRPUU4g7RWMbK7LbBHKnFTZzWinfYhiOOM1EXIOais9SoLUt2s4Bw1TyIsnNZpOyCroyncW6sO1U5LfYOlbWSaMr6FZowTUEsfzetaN8qsjWMtbF+yhBXOOauk7U4rBRaJm7srks1Qy5HNOTfQpMIhk81K0WBmud7XZY6NNhyRVhU3CpSuim+o+NcHmrBf5OKcrWIK4JL5omGe3QU+VWGilc529cVi6icc81dOXusf2inG3UmmXE4CGtqd+Y0lsZsjFietULx8Z612QetmZsteHpM3I5xzXpGmyfuwa357vX+vxOSsjRWXIqRXyaTlpocqJuTSMtQ5tiIJUqIDmuKom5GyNCwTkE10GndR711UHymU0dBaodtWMcc16Lloc99SKQmoHPeueZSI896ehJPXNYqWpZo2kBk61ow2oH8NdMYaamb0LK23HIpslr3IGTVNWIUiAwheKYwFRbqNXZHwTTtmRTW2o7DCMVG+Rzim1cCPfg00vmo2LaG8E0MBmlYLgmA2ak83HerjKwmiNrlQeWpySCTpRcL9CUdKXOT1pLQCZGqVVJ5prYjqSChnxVoRXmn+Y81Subj34qZMpIoT3HXJqjNc4yM1HPYtIpT3XHWqE1zk9TXNUnZtI0jEo3Fwc9artMSetckptu7NokiSZGc05Tk81cXdESJ1wBzQzDtWrSSuiE3cI8571ZjjJpRempUvItxwkYqzHCe4reCexLYk0Hy9KpyKVPNY1o2HF3HRNg5NW4ZAMHNKDsimtS7DeBe9XY9RQHOeK29ol1M+UlGqqATuwKVdVYthT+lHtLvQlxLEF07jOcfjVyKZ1I55FaKKtdkSLsFyeC1XoL0HBJ+vFVB2VhNdUX4JllX+tOIBINJ7EpscF7nmngE1KDzY9IxUgt++OKcXZksbLaFlxjgjmuD8f+EU1aykjKBtwNdCfQmL1Pl/xp4NudBvpEaIiPPyntXM+QU4I5reE3azLk7MVIyWzior2LPatIONtSZLZlHyzu9qJRhetaqSKlroh1rJtYZPStE3MRUHuKpyTVh7WOx8H+JvLhMJfA+tbVxqZcE7uvvXi1oWk9NT1aXvRuzLubvfnk1k3OmS36MURmxzVUpKN20TUV1Y5O8hk067BIKkGu20XxPJcWKwtIxGMD5q6sRFTgnYwwspKVjmJm3A4qrgjk11OOqOMbISo4qrJEX696dPdsbiVprIkllFUJY2Q/NWsGm/Mzs7jC2BzSBjurRK71Fy31HpLtarKPuFRNaXNoJit7mmMOPSudR7lpCogxzUix5IGOtOXkGp23gXwq17cJM6+4yK948K6CLeJBtGeK1oRvK7M6jO3sbQRpzgGtBeFr0UrnLLVgzE9TULvwaoVr6EEkh3cHgVWmk7mpTGl0Kcs1Up5uTjiondO5SM+4ucZyayb27wCSelRJ6GqVzntR1ILnByfrXPXWosXPzVyVH3OqlCzuZ81yXPXNNRZHGcZFeNWk23qdi0IrhSnJ7VUN2y/xYojDS4o6oje6dx1puDKKVS1nY0tbUiW3dJMjpVtScDHWudau5unpcminYMATmrK3JPQ81jJamiSauHmMRktTluRtw3Wm+VuyBxvoiNwG5HU1E0WBnvRy66D2RTnRi1Q7mRhup8t0aJk8V1z1qdrnK8VTp7MTIfPO+r1vcEL1ou9w5U0OlnLGrEG6SPI61i0lsTLREbQPu5FOMBIGRUKe6Q2wMjRg1n3DFmzVxirXLikxYmIHHWnuuVyabjezKehHDJ5b8mn3V3heDTnTa1ZEtWVBeFm5NWEnIHWnruO1ga4J7mljmxzmml1EWI7wgYzViC6Ibg0pKyszOcS6lyaUylu9c7lKxk1YAcdTTvOAqJzd7MahcR51oiugpwTV2W4cjRN5oJBqUS4XNEGzNp7glwA3WrsUwZeTXTHYzmiG4cAHJql53OO1FTcIajJMMalRPlzmuSbdzdbEkakGpOD1qlJPYiXkOXGKYzEGqkk0T1GnrmnJLg8moi7D5bkwYFaayjvVyX2haxYo+UZB5p4nYJilBW36ik7ld523YJxSTOGXA61XxXsDRWEeTzQ0Iznip5rh5FmADbVlUyOavmT0BqxFLHtNV5CDxUyiOOqERecipwpbFc8mrbGiJ0jAHNSYA6Gs6UL7ktsY/wApzQsw7mtVHWxSV0AxnNPOcYpSStYCheRNzisa8icg5qKV7uLRpdWMmYtGcCmpE0oya742BajJYCO1ZeowlQWNKlLmlYipGxX0ifyrsDJzmvRtIvN8KnNd0lZHPUV0bNtJubnmr8SFsVKfPsc0lYtRwHvT2gpqNzK+pVuIsCq6p81cda8Z6GsWaFpxit3TSSR61tQepE0dHagBR7ipyOK9SS905epC681XkX1Nc0lpc0TIW4NTWwy4zXM07l9DcsVGPWtGMDFejFaI52ncnBUDOajuZVA5NElaIrMzp7pV7j86ovcjJwawc90bRVkEc3NS+d6UKWgPccPmFI+Np5pt6aE9StJ61GWwOtJtWLvcYHJp2T3NF00DF5NMk3KCc1LVgMy5d9+ATVu0uVjGGOTUQ0buJ7Fz7Wu3OaWG5Dv1roSVhW0LkXPNWV6VViXuKXAHWq09xjvSsFrlOWUkdao3E2M5NQ0WmZ1xcj1rNuLwdN1c8nbUuK1KclyT0NQu2RXK58zdjVIrSDNV34OOlS4FRATYNTJL3JqIN3BoesxqaM5rVSvuxNFmHqDV2Ac1tSsyJXsX4YtwzU/Cjiu6MV1MXcrzyVQmYbq5q7SRrBDVx3qQOVHWufSxo+wvnH1pVldj1NY1aulhpWLMQc4ySav20TdwaunzSjcznK5qW8ZFaEEWRmu2F+pi1poXkt9/bFSLaMCMZ9a0StsZ3LsCSKO/NXIyTwaJQFzEyoe9SomTUIHYnWPvUkYweam9iWTeWCM1Tv7FJl6An3Fbxd9SbnlHxP8AAcN/Yy/ulJwSDjkGvmrW9Iksr6WMgjYe4rbqXvHUzkYKcHrTLlVlFapRtYNUrmdIu1j6VWlJFaQWg1FtXIfmX2qRWbHBp310FdXNXR7vyJlxXXW0zToOtcuIWvM+p6uGd4aE8Vi0kg64rtPDGkwJEGkRTkcgjr2rzq7aijXR6nnXxU0uK1vd0HTJz3Heub8KzF7tIyfvMB+td+Eb9i3I5bpVtEMDnuaUY9a7GcnmNYbuM1Ey56dqIStuUmMY44qrcW/m5Iptq90DWnMzNuLd0zmqxBB54rohO60MnLQXnqDUscjDqRRJ3RUXrcl840vmkrWTRdxRITyTW54X0WTVr1PlJUH86duiNE+p754I8Li3iQso3Yr0rTLIRoOmR2rqpRtscVSRrQpgVI3yjmuhGLZFK3eqzvxyelDeg1rqVnkzzVWaTIPNJsGUJ5tueaoXFz6nrUz11NEjLvLrAODWDqN9tyOvvXPJ2RrFHOaheAg85NYV1Md/WuGrJ6noU1YS3JZvmNaCSKqYOOK8ua5pGzs1oUb07ycN1qtHZF+c0cztZCvZCSweWMEimxKM9aS1Vy4u5PhcHNMZ1BrG3LJ3N46jldWHFNZ9p61i1Z3RpHRjvtPy4PWozLzmq5GndGiJ4puRVjeGXilK9rsmaKzoOagkhB96akKTdiu6beRSRszMATW8G+UUXox5jIOakjlI4NZ1GaxZYik3HmtC2mCLz0rGS0uRNaWHNdr3OKcLhCQOtTya69RqNgYJKOMfjVK6i280STS1FG6ZEnNOP1og3sjVkEqc5zVW4JI+8a3d2rGbbK0QYvkVcXJXjtWkGnGxTBSd2DT2yoqJNMUtEMExzzVi3uSCBmpk11Ho0aEdyNuc1PHKTzmsWkjFom8wYph55rOUbse2hFnLHninBSeaz5rmjJozmpw5K88mqhJ3MahF8+7npVhLkx8CuqnJ7MwshJ5srnNUGu9r4zWsktwgrsHuiQOasWtyWXnk1x1YtK6No7WJ1uCuSTSfaQW470QatdktakpnKrwadDKsnDda0tdEWuPkIX0qHqeKxkrIcWSByPwpDIWNWpc2jCSJYQTwTk1Y8risZSalYiRXmg+bjmk8kgc1qtCeYidfSoWU4NErwVkXEktw6HnmrglIFQqly3ZsRnDrVWRfmp3dxR0dh6jYMmpUkyOKnl5nqPzH7yR1o+Yc5NEE07A2iVcNH15qm7FJOueamreMk0OD3RPE+RmphyOtW2mrie5DOMjnNY1+uwsc1FG0ncpvoYF22ZMYFaFjD+6ycV1NJ6ovoRXMGScVk6jb/Iaij8V3/X4BKWljBVGhuAwz1ruNAu98SjNene8Ucslpc6eyckjmuj0+IMoJ5qKPxWOOrtoXvLwOlIY63lFIwTKd1HjNUtmGya82t8RtBly24wa29NPIzWuHSbJn2OitXygzVsEHrXqrVWOexHJgCqspA5Fc9TRXKRXbk5zU1ufmrkUryNOhtWjkDIq8kh6k16SkuVGEkEt1t79f1rLvLxs9ayqyaLgkZs905J9KbCWc1xqd5G7skXYkOMmpQMc10Kxi3ceGIprscU79BFaViO9QMWZuKyk+hoiWKI+lSmI1pAzuAU5oZcjmqXYTuULqMAn1rNmZ0bINYTk02aRt1I5NQaJPSiy15BLy4z9adOr76RE9DoLXV43AO4fnmtCK9SRcg8V2adDJMhu78RA4aqf2sScg1HN71kWiNpuDzWdeXSrnLc0p+Ra3MDUr0gHDdfSufmv52m6mvPnK0rM1j3L9tMSgz1NOkl7Zp2X3logklA5zVZ7lWbAOTUSTloWhuCTkGniXHesObl0Y7XJonzVqJ+OtNS7ikrFiJz+VX7aTcea6KdRRsZyVzVt3UrwelPdsZNektVqY2sUpieapyHJ5rz6yvobxegzefWjzCT1rlcrKxVkSRoX61et7Q7uRRFKWpMmadvadOMj6VoQ2vTiuuMWlYxbL0UWMVet4iTnFakN2NGCLjJ71ajhDHNaxfYykWo4dw5FSC32irirmbdmKAAeakRu5qJaM0vdEwcEdRUicnOakCdATxQ6ZrRSstCGYXiDTlu7dgfTH4V8z/F/wpJp1615Eh2t97itopuJdOW55XPCSSSKhb5R2q4NXHcqzR5BNVJl281rCy2Gmtim0oDe1PWVWraMepMi1ZuPNByOK7DRbsbQS3Fcte/Loejg9dDfhv4o+SRVo+J/s6Ha2AB2PNeXOi5aHc42OI8YasdQ3Mcc+pya5PSL42epxsP74/nXo4eDUeQ8yUn7W5oZweTmhnzXT5syI9x6g0pfim13BIZkk5NNJzUX1KtfQZMiuvSsm7h2HNXT0ZlOOtir35pwbtXRe5N7j1yMmnrk9TVOz1Gl1Lmm2j3lwkaqeTivbPh34QWKKMlMYOcnvUct5JId7RPYtE01YY1OOldBbwBRz6V2QVtjlkT4xyKRzmrWxCK8zYqrI3BJ6UO1rjuVJ5Md6ozzEUk11LWpQuLge1Zd1cgZOaxZavsYuoXRGcHrXOaledctxXPOXc6qaMKefzG61Tm3A5rzMTUdzsirIhWfaTz+NTC6JAGTXJJu+hrbqJ525sml+0FRwaFNR3Eo3ZVnneQ4NLExqeZPY15bDnLMevFIyk1FSVzSKAKQepoBJJzWafQ03BlIFA96OYbXVBkqe9SJMV5zVOS5dRx1RIJg3U0jDPIPFZJalKNhjQ76jaHy2zitFKzsiGgE27jFOEWTmpqrXQpKxPGFU9asbht68VKhdCk2AQPyDUMoZG4NKLYRetmOilcHk0+5csnNOXvItvUgiBzSyMV96iC94u92MPI5qB4ixziru3qQ31GCIA1OvA6cUlLlDdajXwCMCh2ytaLUmaditI3PvTI5iG606kOxVJdy/BP0BOa0bcliOeKym7K45xsW+i1DK5GetSmnscyK5fa2c1PHKxHPSs5Jbs2voWI3xjipTIAM9KhRV7mTE84DmmtIWOa1lpsZ8t3cU/MvWqNwgBJzV3lYkiVsnBqxETHznipqSZrsXI2WSPrVV5TFJ14rGm3qpDS1Y4agcEGkS7IbIauim1cFTsWY7ov1OatROOCaibT0IasOMg3fWnAd6mKSIfcPO2HrU4uiwquVPVktdRDOOppstxxxVpJ7k8pVMpZvSl3E0p23ZWxPAvPIqSbgGsXFWuVfUiiyW5qV4+lc8ubmsaMR0BXmiFCT0rpg0ldmTGSuYmpftIZcdKaa3YWuRPdPCvBPNU3vGLZP51bcZxszaCRdtbkSgAVoxJnGTWDVtERJWFu0VYya5rVZdmfWtaVLkVyIu7MPb504z61tW6BYgo54pSm4ux1W90ZOo545rLvoSc8UU11MjDurXBz0rX8PSlSBniu2nUSX9f8AiasjtNNOSK6iwl2pzV4dvmdzz6y6F9DuGc0412PXc57alW4QsapSRHPSvMxEVzG0Ca3Ug81rae2GBzRh3qORvWs3AI7VcE3Gc16qfunK9yKWTNQua55u65TRIYFJ5qeFOelZKCTHc07ZtvFWllxXZDYysV7q5xmsuabeTzXPXmaQRXCljViFNvWsYxu+YtstI+OAalVs966EZslAzSNHkUyditLASc0qW+CKytd3Zd3Yl2gduaAcnmtlZaEtC4HXNNYqPrVXQrPcp3CqxzVCeAHJx1rFpdSr3MfVoysTYFcNqerT2Fydp4Hr0rinFQmnYU1dWLul+NwNu6XB7gmuy0TxIs6j5ga7aclJmKunqac84mG8Ngd6rfaNmeabajK5ulcjlvSQeaw9XvJADyaitN8q5SrGF9pllbmrEcIfkgVyztLTqVFEmNtRyEnk0m7JGqMvVbqSNcJnJ6mqOnTSyyZbpRGemoc2psKcLzRjnNYVN7lokSQqKswy5xUKV2KRdixirMMoQ4rr5UZtdC3DeAGp/tnGS1dcK3T+vzJcXuQSXIbvVd2LE4rjrvW6LirCbackZzXM25Mov20PPNadvGM10049jOWrNW1jGKvxR5HOK7YtNamEtCzFESelXIYcckYz+tQo9gb7l+CIgD1q3DHg9DzWkVuQ2WkjOQTwKkKEDJOK0hpoZN66lWdiD7VEJSB61lKWprFaEiT1ZhuBmnbUTRbjn+b0xU2/eaV+jIaK91EJIypGc8Zrzv4h+EY9Us5UZA2QR0rog7uwo3T0Pm3xZ4NuNGupAY/3YPGK5C4tmjbmtOXkdkXJt6lSQjpVOYbs9a3h7t7jjsZtwuxjxUSSEHrW8dgsW7WcLIO9dBYX20AZxXNVTsduEfvml/aPGd1QTX7EdawitLHptN6sy9QPmg7utc9JG0d0uB/EK6ISs7M8vFpqdzZbjODTGbn3pybMb3EDEDrSFh1pSlfVFR0E3+1MdvehbFXsN8zHOahmRZe2aNUyZRurlCe0KnjNVyhVua6KctDFSJA3PSpIl3MAO9atEtu56R8OPChuJUmlTr0r3vw3pCW0S4XGB+vFFJNyuFR9EdfaRbQMDJ9fSraZC89a7EraHPJdBxbAyajd8dKGK2hVmfNVJZCQeeKkNyjPLmqFzLjPzVL7Foy7mfvWReXG3PNZSlY3gYGo3w+YE1zGoX2SxzXLUdjrpx6GWLwh6dLc70x3rzaqTOpFcKxapxC7DgVjzK5T2E+zyg/MDigq3ocVz1JJmkWmrIBGCc4oxtNYrm3NY6jwy4zTGlAPWiWpSTQCQU0DJzRT1bkX1JR8w+YUmzvTdk7gI68ZqPnFKV3ohRfUZv2nqamE2BXRpJFKV3qPSck1K7B0rDS9hyRTJ2t1Io+1YbFa0433GiRZ93NSefkcmod1sNokjmAHWlaUHqc5qEpEt6io65pzuHGKabvZisx0UeOac8Ic+9YvmuNMRrJsbqYIDnkVcJXIcrsjmgxziiOLzOKdgi2LJbFRk1A8RFTGTvYtEf2YuelRPZFHzzitJVGkClZlq2ty4461et1aM4ao5nJ2HKV9y6JBtpsihhkVPI4s507FcxjdzVi3UA4NRJ30LlL3SydqDORVeaXvmnCGl2ZRK4mOetSLc4HJ4okaJEqXA28HrVa6fcetVB9yZR1IEbDc1dUhkqavcOUjV2UkUpBcc1MWvtDIJlxUcZOcda0VktC4PuXrdG4PUVoROMc1jJ3ZNTUJCFYMORTjPuWqSaVzGxWnuGBwKkhuCw5PNUmo6stQ0JjJgcmm78nNRz31IcNBpbDVIjA85ptitcu2xUYyc1LNGpGaTTZEnZlZ3WNu1Rm+XdgkGr9kty1qiWNhL3qyqqgzWc4a6EvsUrpDI3FRiEr1old6ItPZE0dv5/BHFMutKGMiskpbr+vwDnsyO2t/IPpVtJ2B61pFy3ZT1JtxmXk1jatYlmJ61spu39f8ABMVpIxBbtHNwK0onKx/NUS97danVzLlIGlLnpVa4XJJ5q4KxGxl3ygKTio9HnCz46c1rTi7siWqO60qQMFOa6Kyk6d63pSd3/X6nDUNeInGafursvdHN1Gsm4/WozbeorkqQ5i1IBDg9Kt2oINc1NNSG2atvIRirayEjrXqKWhlsLuo27qmSW4th6oBU8QxSaugLCyEd6WS52L15rS9oisZ1xdFicnGagDluTXNLV3NUiZBkVOqECqVkhN2JBntU8ecc0IGyZFNTrHkVoot6GbGvEO4qJkPJFJqw1oiNgV5qNnwaV2wI2lPY1EZTnrVcxSdxjNmoJGqGSV7my+0Rkkdq4XxX4dyWKjmoxNP3bo0pyT0Zw0ulzWsvU8Hit/w9qj2rhWauelNXslqc9S/NY7S11dXiGXGfrStfgnlq2lK6sdC8xftAYZ3VRvbmPOGaqkk1qNsqqIm5GKljQVzygoyuioCsnFQyIMVNSJSKNzaLKORnNRx2Yi+6BWCvuNbk6RlqmjtSe3NVbm0G5WHNa7aRVKGs+RrYcXcsxzYHWpEl9TWvOxNEnnEd6Xz2Ixml7VxehSRIhJwasIpai8nuS2WI4NwzU6W3I4rRU0yLl22tiDgVoQWxyOK0ina5EpGjaxOOMda0oYGODjpXQkzFuzuW4YtpGR+NXI1BxVXaVkJluIZ71ch4/KheRL2LaqCOOKHjytaqSt5kO5TuYD25HWqEimNuc1lPe6NIPoAck5Jq1FJ0qY76jkWkkAPBqZJfU0+pCRLuzVLULZZ4yrDPetI7kWdzy34h+EI72FyI8FskY9a+efFGgT6fdSJKrYB4PYiuyUJNKS6Fp6aHMSWwYkVA9rtHJ4FPmugcjLvY1JNZcox61vTvcELBPtfFbNjMWYHNOe2p0YaVpmoku4Z3cUrSDHSuJpt6HtXuiFyG61Qms1aYPjoc00mnc56tPm1JH96jdwBx1rV3eh5ck3sR5zTS3ND90uN+om+ms2cjNC1VyrW2GkZHtTSCvSq5ujE07XEYZHNVZrcEk96cXZ6GUoPdFZkZGrf8J6M2qXqAr8qnJOK3k9Lkx3PobwT4eFqkZ2gZAr0fTLby0XI+lddNaXMar7GrGu0VJ+NaoxI3Pqahdsd6QypOc85qnPJtNDfQpdkULmT3rLup+Dms20VFGJqOoKik5xXN6hrH3gWrlrTsrnXShzHPX+p7gRmsOeZnbk5rgqVup1QjYjHPJqVEJNckpa2NIRd7l6CFSMtVuEIO1cEr82h0crHvhu2aZ5CnJIGKclFxFGOpBJagc1WlTnGMCojK+hqQOhXpVeTIzRzX06mhGrMCOvNW4ietNTa0RViYMDSF+ah3sUog3I9KryNzxW0VomKK7gFLNk08rheKtRfQGkxEfDVbjIdamrGy0Ka0IJYMtUZsz15qPaNLQBGgKc5qvK7ZIya0oyb3HfUatw69SanjnZ+CcV0cvUTZcgRm71OY8c1yVr3bW4rk8TgAA0F8PnPFcybS1BLUmS6AGCajeVS3Heqashclm2NmdGTFVI3McnFbQ2JimnqWzJvSq7pk9Kxl8dyhyAIMmiXDjgCq5b6sXmLZ/I/arjlc59aht3uFrjhyvalXnvTdS7MZCNHyTSKwQ9amUR3bQyacqM7s1XNxuHJra99BJOxEXO7OaY0zZ5NK+pqhUvAvUmg3e7pT5NG7jtcckgZsk1cjcAdetZSvZXCw5yBzTlb5MVLjfUhxIJM96akRLj0qpe6ghpqXo5Ni015xnPSpgrxuFtboQXZzjNWUbC5zUychuIxsE570RfIc02rkq6JtwI60xpCg601FC3IPPO7J6VKZ8rn0puS3BxsCap5fBqwupGUdavn5b3/r8RezIpZGck5JqqFdn71mqz3BxsaVqzRgbulXmuVZaamnuZyj1RXacbqjmugBiq5VFgok9lcgDJp892pOM8URtsxOLTIJJQVyKiSfnBqZ6IpFuKbimXDBvespT6BbUqPZqfm2jNV7mEInStYx6/1+RnzO5R81Is9M1UuLnPet1BM1SMq/nJBHc1U02QrdVtSio3YpK6O90eUvGp6V02ntwKUHaVkcVVW0NeKXjrUqvk12tpaHNyk8YB5NTeVu7VHJdCbEMFOjj2mudwcdhluIHFWYzxk10xelxSJ0wRT8U2yNbgGFSJL70KQ7DmnwOtVbi5LClUloOKKby7j1qeBSeTXNDfU0exciWpgM8V02VjNkipiplSlFdRN3LMQFTDBrRN3JBlqJ1xSkguVpTiq8grPctO5VcnPpUZbnrUq6ZXoSKoPenCBdwPatlHmRnJ2JGVQueKwNZt1lz0p1F7jQQvc4zVtKUknaM/SucuLWa2fKqRivKacJcw5wvqOh1C7jXAyKv2uoXOfmzzXRzaf8AqNiw+syIpABrA1PV7kSZBI5pp3STZci1o+tOcb+1bcesIF6iicPMI32J49QSUdTT87/AMaiS0ZqhGhbGaaEA69aws7BuSpGOpqZcVcY2YrXByMVA45pVbX0LirEe4g1KkwHeuR1NbMtxuPD7jk1agjD470Qk5OxEro0rexZ+1WVsGXnFdjptK5k5dCeGALjNXI4c809UiWXrW1zzjk1pwWZ/wAiuiC01Jky7b2x4z1q/Fb4yOtaxsYstx2+VBIzT1g29auadiVLWxPGgHIz+VWogcYrON7FNItRHHUZqYKG461XUjuH2cNxgGqN7ZbRkZq3HmQk9bGdLFsPsKFbb3rDZ2NGyxFJyAep6Vaixnk8nitLEpalqLJGCaV4t1VHchmB4k05Z4iCMke1eO+OvByX8L4TDjODivSpS5o8rBaM8N1jSZtNunilQqVOPrWVcJkE4NYpcqsyjLuogc1l3MIGeKqDY2ii3yGr1hNzjtXRJO1yqSaZrxTbgOSKk8zC4JzXIz36fvREBBNNcjI5qOfXUGVTJk012yPetbs8UjJx35ppziqm77jvbQYM5zTTuBJ9aq6NFoKpOMGlJ4wDzSvcd+gmTS7COeKG7E9LMI7I3MqqgJJNeufDfwl5MUbFOTgn+dawbdkZSSirntGiacIlXjAwMV0VvHtXjrXfDRHJPcsrmhsY960MyFyfwqtKxz1o5kNFSZyO5IqhcSjmoKM25lIyM81jahdmNGJrOTTZql1OJ17Vz03Hg9q5S+1RiSxc15+ImldHo0YpamUNRaWTBbIqYhnGa4Ztbs6JJEsUBfirkcDRgZrkqSV7Fw8ydRt6mhZPm4rN7aGkXcsRueppZJ8CudSVylHUgaYtUMo7mqSSTZZA6k9qheLjnrUXbLe1hggDGpViKnrV+8tiUyXYeMU1kJNPpqXFj/LOM4qvIpz0qlLuNbiojYzTiOORWylpoJ7iGMtzU0QI4zWTfVl20J1i3EcVLsGMYArKUVa6Mk9bFW4X0qjLESScUqUrGhH5BdumKmhtWBzXVKrYasaEClB0yadI1YSl3BoarYPtUijdWU3poAPHjrUb/Kuc0uZyFcYrknmnCPJzWzk1qKW4/kClBwOajfRh0DIzUcpIpyi3HQzbsEbheppz3Z7HFNpcquax1FhvDnBNWvO71HL1JqwswFz16VUnuWDUN63RlFa6kf2gMuCaYrc5BqtbXNLWHHj61BJJsNRTi9wbT0KdxN83BqaAs65yTW9N+60ymvdJVYoc9KsJc4H3qUugk7omW73gDNSpIc8GpUUS+xNGN3Jp5QdeKmaW5jfUQE7gCfwp8kXy5BrCSexomRCElqu2442miN3uDl0HNEcmoZfk70JvZC5riRy5p7J5gquZLcT3Gm3Pc01oWVTg1knZ6i5ijcBgeKktZWUc1pNaXubQasWxLkZqa2Kse2a5vhZU0Xtm5OKhYFT3q6bdtDnT1sVnlIbHrSOM9avnepo1YdFIQetJK7EkjNRGb3FZXFjkO3k5pA43cHmqnNkKOpbhf8anCb8c0KSkrMmWjuK6YFZupKxQ4rpT5dGZJ6mC8Ejuc5xTZLfYneiFRt2R0yemhl3inJqjDKIrgc966oasxb6HdeHpRLGMnOMV1drJsAIxT05tH/X3nJVvsXopycZFXrdS/NXzczMWaNvFxVqOHIrfqZMkMGe1L5FZTiIfHCR1FPANU1oK5NH609mqtwIWkIPpQJjjrSvysojkuOOtQPKWqKjKQiDJq7AcLUU4ibLUbe9TxLura/REMnCY570ofnk1S8xJkyGrMY4zVx7ibHngVBMaUmhJ66lSVgOpqBpFbocmoii0ytMeTVYt81YVJNS0NOg9J8d6VrnHeumDI5SGS7JFUbiQyZJqZzvuNIy7qAMTxWZPpqOT8tcMnyyLSuQDSo842Cpf7Kjx90VCnzPVA4oY2io2ciqlx4aSUfd/SjRbIpPoyBPCeD8qmrUfhRyoODUc9Qdkti3a+GZgwG04rbtvDfyZYYPvXTTg5R94m7uQ6ho8kQwqn8qy1sJt+WUilODhsEZEv2Vx2xSGMqKzlJtaGifQibI61TurtYFJJ6VjJu5aehk3PiCNHwGFLb64JDweKxnTTjzJjTvoakFzvGc1t6MDNKvOa5aVZupyr+vxNJx92522naarRDgZqxLpg25UHivp+ROGqPN5tdyq1iR2p8FsVbkfnXDKHvWNLmtZWvGa1oLXGMY5PWuum4mMpdCzHb88DnNW4ICfpVK1yXsW44cjmnNDnkD61ctVZGfXUWNOeasJGOvb+dYwi09S2TolPVDnJq76kkqcdaWWMOmMZAraLVtSTLvLQHJAOKy5Y2j6gkfSuSorO6N4tDo5GPSrts+SD29+tWpX2Jky9Fk+1T7cjrWsW0ZvyKOpQb4jmuB1u0V5WQ/Wuug7NMDzDx/4MS+ieeNP3qjPTrXj+p2bW8rxOm1lOOa0rx1uPexkXNvweazZ7RtpJFYobdkY15A0bc0y0lKvXZG0oCjvc2IHYqCan8ztXNNLZHt4ad4WQhcjk0jSFuc1i49zdyRWZiOjVGXrdJnj20uIT3zTSxxyTVrVDS1uLGxzjpmldgRxWc1qN73GD60nJPWqT1aKjLUdynenou4fWh9xW6nYeBvDxv7tZWThSMce9e7+E9GFvEmFA4H4V0UVcxrNnaWcHlouDV+P3rvilY43rqO37aYxzTEyKV+aqTORnmp8gRRuJDk81mXMxHOaRS1Mu6n5Jz1rmNfvWWJgG5+tZSOiEbs871e9czFtxNYd3M8gPNeNWfvNnowsiKziJbmteEDbg81hVd9EbXuWrcAHmrLSp/erimmb6pIYHDd6khgDnOaHJxFYnki2LnPAqozEtzWCtJlwHqmec0SQ7hkGpTum0VsQFCp+bpTJE3DIrVJNIq90JHBmiQbTRzXlZB1sKkmTjvUyoSc1nUXM12LskPI25zVOXqacFdsB8OCPSnSQgcg5qlO24+pHkr1p0b4bOKrSSNUi7EcjNO25NOCjLRHFUdmMlgOcmojbBulYyhys1pyuiM2hSnxIc80m1ubR2LUcW4cHmoZ4DycmiadnYy5rS1IY1INWRGcfSiMv5hyeox2I61GTupx3ZVtAQYPNPABPvVSTtcmW4rKccU0Ix60J6iuhdpVutRTnPU8001sZz7lNpth9qFnB6mrlHmRcGPDgckirMd0GAGacVdhJ8wjvgk5qCaQnms1HoSQh2zyasW55+aq5UyraEsqZXIqpNk8VMPdbRNio0OWz2q3bsETkiqvqarawsjBjwaiZyOp4rRLuRsIk5z14q7bXHPLUO6WoOJdW59xVgShwMGsWiHHQYSd+c043HOM1jbdlJXJoGzyeTU+/byKtWsZzWoqzMx45FRzoT2rOokrNEqyZWPBqzbswHPepdnozRrQl3ZODio5nGMZqVGxHLcpyxljnFOiA6Ypyae+xZI8Z20yGVo5BnpWbWuuxcZX0NOK6yvFK0gPeqp21Ri1rcp3BAOc0I+VpOPKn5m28RoPzZqWRcx8GiKdrEMpXErRHFNglZ2ya0i1y8xqkrXL0cxyOTVlLllFZRlyasylFMkE+4cnmlkiEi81s5c6uYTjZ3M66t1TJA6VkXjlQcZpUY2kaRlfcyp080dayLqIxSZBPWu2nKxMn0Or8LXRKqCcn3NdjaXBbHNW5aP8Ar9Tknq9TVs1LkZNbloNoFXQV3qc1Q0YmwKsxfNXUvIyaLKR5FPWKiUL7k3FMdRumCDUSV0NDlJAprOQMZpX6BYgd6geUg9TUOWhoiMy5oBJqd3qMmjGOtWUfiqiTYsRNkjmr0OAPSto9yG+hIXwOtNDjvROzegkiVJPWrEc2O9VCwNDzLnvUUjjB5oaJ6mJqt80RwPwqraXTuCSTXP7R89i0iZ5i3OahkbFFRFkJnOaUMW5zRTk3oVJWBj8uc1WmcAdac49SVuU5CCetRMg61zSSe5a0I9hzxUiwk9qyv72hRKtsxqVLEuRxUyXM7A2XrfTM9qvRaWvHFdVKkmjJ1CzDpoBHFaFtYrnkdvzrenTs7EOdyaXRI5lztyTWfc+G0BJCiuiVOLWpEamupmXXh3AyBjNZV3orJ0Q1wzp63R0xmnoZV3YugPBNcpr9pdO2EBPXiuCpTknc1bVtTnl0O6lly2QM1r2WgtCATnmuavWnayYU1d3NeKDyxWzodyIplBrzKdfkqcx0yTcGek6HJ58OQa2DbeYucV9vSlzU0zyH8RC2n7s8ZqM6eUbOK5pxXUrmLVpBgjPeta3hyKUH0RMi0kJxzVmCEd62StqTctIiqOBQVH61baJtdibOelPRNrc1i3Zjv0LCAE461KF2itGibCHinqSR1qGxjZIN5xnH1qpNp4Y425rRLTUWz1ZAulgDAH9KX7L5XPXtVwh1QnJ9B0bEY9vWrKSgjmobs9SrDbpfMjIrifEVv5cxfGB710U5a6COdu7RLlCHGR6V5V8RfBBIa7tkw38QHeuyUeeARfQ8xubXyCQw5HBqm1sJQcVzqASZi6pprDJxmsSSIwuSa3pqyt/X5Di0XLSf5Rkk1cVz/eoq2R6OGqJKwruQeTUbSYBrnsdb11IiuO9Nc5rSTfMecxjcikycVfQaFGaDwagpW1uIOTT8YOTVta2C4rc96vaTZteXSRqucnmsr2Q2j23wD4fFvFGccgV6lpNnsjGRg4Br0KC01OKrLVm1Cu0YPpUq4FdKOcGJFQyNx1pjSIHbiqs79aTBroZ1zKMVlXdxg8msmWkYd/dldx9a5LXL8srZPX0rCbtG51UUjidQYs+OuapNAX6jGa8eesjvfkXrHT93b8av/wBnMoyMGuSrJpm0Vy7k0dpxz+NQXNqVOQa5VUd7o0uMjifNTrK0Y+la2BTTdhPtTuCCajL7mzUzgo7Gia2HLIVFSifA5pQjG2o276Eckgf0qJsgVKa3iVa2ggkK00vvOTT5balJXFCY5Uc1OhI61nzcqs9yraDn571Xki79aV7aijuAiwakA4okn8Ronca8e6m+WVOTRCNrIZNA3OCatjoDWiTg7mFWNxs0nyVDbvvcjrTqapWJpKyZfjtfMXkUCwwc+tcVR2d0WpEq2uz602S2yD0rWKbIctbkH2XackCpBEpHvQrPQG21cr3FvzmoPK2cmqSaHGdyJnwcUgfHtWkZ3iaWLEMyHAP51ZaJSmVNCWtzKd0UpWIY4qrNljkVaaTDdlK4DrmqJneNuTiuim1cqNiOXUCq53UlrrBR/mOa7KVG92RKWmhqW9/9oNXGj3rxXDWhySHL3bEBXY3NSpIvtWUXzLQcdR5kytQswJ5qXFt2NLaENwSi8VWjlfPWtYNK90EddGSGVuxpwbcnNau1iGhnf0qWNyhqJN2GnrqTfaiTgGpobtg3U4rPbQtxLiymUZzT1TJyTWM3bREbaEsUu04NWUO4ZJrK7TsyZLqSxDB9akI3HBqX2Zk0Q3EQXmmpINuOlKV9BJ3QksmwbqrGYu3PNbOnLcqL6snyNnNQlyG4NTyIcbbkqzblwTSFVPIFKSsrIpbkqOUXml88bTzzWUE7iaK3mlpOverceCMUVP3kkjVqyB1CtxSMTt55q4u25mytOvmU+KDy15rOTS2KuTKPzpzEqvvWMl1YvUdby72q9HyK1pJ7X0M6hWvIweM1i6jFhcCunl965lF9DEuFK5xWbcqTyRXVBXKk9C9oFyUlC5713+kneqsT1rarLT+v8zCpGyudDaYXFa9u+BVUmkjhlEuxP71et39a6IshotpJxUitxVMmw8DIodfWpeomM2d6jkQ1m0NFWc7apySVzyvsWiLf83Wpo2z3pUpXlqXIso3apVNb2s9CCeJwOtW0m461cGQP84Gk83Hem9BpD0nyalWX3ohvqLYk8/3qN5MjrTk77BbQzrq3ErZPNRpbhV9K53FuVxajWGBVWdu1VJ2RpHXUrFwD1pRPjvWcJWNGrjJLoAdapzXoziirMFHUjEu88VIkTMa5JSdy9C1FaEnpVqKwY9quMEzK5bttLZj061oQaSVPI/CtVSW7RMnYtx6fsPSpVtcdq6IpIxehNHBz0q3DFg5OOK2iiW2WkwRSPEp71o9CdUytNbA/Ss280wOCcZrC2ppGVjHvNGXb90c9qxL7Qk6lc1z1KSauzoUrmPPpaR5+VQB7VVe3CDoK+dx3xaHTTK74zUtjkXC49a85KPNE6JfCepeEo2aJOOg/OuuitwRyOlfd4VP2aZ4s/iHi2wemaZJa5J9ac46CjvYiFuUPpV21X3FYU0kymtC8i1Ki4rab0IJQeOaeoLVPqFhwQd+tLjmlJC3Ho2OKkJyOtPmsNDC+O9PifnnpilGS3E0WV/OhkDHJrelJWuyWriGMHOajeIH14rfRLTYizITaZHpVeRSjHtWM4K12i79gEmVwT7Vha5a/aVOBzjAFTCWozj7hGhchscVn6jZpdQsGQNx3r06MtAa6njXxC8KPYTNPFHmNzzgd64TYyE45qHFxnZDk9Sleh2ySKwr22354pS1fMhWTM9Q0LVfhfcuRWk9bHTTlyyuOdjnqKhkkOc8Zrn5W3dHoRehLwRnNQk5bFPW5xW7jlX1xRJ8veh9i7ibwRScN360pRaBPQci45NOBBNJSe4mrj1i3tXdeAPD7yTCV05JpwSm7BJ2R7n4b0oRxplccV11pDtWvTpqy0OCTLajApTwK2uRfoRsxqNjTErleV+DVG4fCnmpZSMy7m2g5PSsW/uMKSDWTNIo5bWNS2qeRXGapfmRycjHauKvJJM7aMLGZGySudxFXEtIiM8V41dtO6O1RtoTIFiHFSm9CpXM7yNLlV9T2nBpsd0bg9c0Kh7u5S7loRlUz3qtPkUKSFbW5CGIPWnBiOabaZaQ4vuXAoGe9YzurWNYoUgY603PHJpR+HQtCN9aIk5okmndlLuWkiyBQ0RU5rFu7BNibeKYcD604y11C4daM+prTlbGAbnk5qURhxn1rNtplCeTsPepIyec1pGblqyZEjJ5i81EsZibOOKht81iEtTSgmAQZIzUxkHWs5Q7mciOSf3oWQHp+tUnbRCtoLs3DJNMChWqFfmuiulhk4+XPrVCbJPU1tfmdyYIgSJmbn9afJakDNS3/ACm3PYrMjqepqeG5cDax4relK61JnqEuH5qEjGeac7EJ20IJvm96yL+I4JHanC100EnYy5csetRqh3Zr2YXsc99TSsZzGwrctbzcuCa48Qk9zqjFNBckP0NVcsh+9iuHltoiodixFMCvJqGaYRuDVxT5ikiZJI5U5xUMsajoabpvci7TI9hB60Bc96cdS/MGz0pdpx3q1pEQRZDc8VaQ+lZSdnqXa7LEUhSrKzfLWT1ZMoi+bznNSrdEYHas5pJ3Hy6Fy2lEnU1aMiqMk0OKe5zVFrYq3FwGBziqnn4P3qqaV9CYxJZJQ0fWq8e7zOKtuyBLuTyEkYzULhhzXPUk90XDsRiVw3FTLIxHzVVr6tmsooJ7wKuO9U2vSD160oJjjEfHPjDVctrkk5NO1lfqaSjcu53imyKWGK5oys2mc7VncbHHhuaW4bYM1soJvUG9RsL+YalZCetZ1Ip6CbH21vsbcTVlpSOKKacVqZSd3qV5WZ25qC5t/MT1Nbb6kuxhXtoyscise7h5Nb027af1+AXE0n5LofWvRtGOIlbjpWzu9GZVttTdtX5HNacEnvVxetjiki/BJz1q7C3rXVS8yHsWkkqxG+TV3uyLFlSMUNzU2aE7gelRSUMOpRuORzWdcMATiueceprErh8tyasxyDqTWVPyLaJ0kyasJJ71srrch2HrLz1qRZ8HmruhWHrPnnNO873ochDo5vepln4pp23EKJ896XzQapbidxryDkmqVzqEcKklgKSQig+spK21XGKGl3jrWEmmzSOhC/FQOxHQ1k2kzVMhmLMODVTyHd8muevJ6ML2L9vb4HNaFtb78e9OLVtdyWzYsNMaTHFbVtoIIyR/9euyhRu7sylUsXk0cRgEj86l+yBQCT+ldbhy6Ix59NSMxeophi5zWNRjQBdvJpwfbyaamN6j45s1Oh3+9aKXMS9EOaIHvzUUsAPFJprUEyjdWg547cVh6nbgAnHTrWNV+7c6KbuzmdQT5jisidexr5LGycpOx200V1tWmfaorpfD3hZ5XVnFZZfg5V6l3sKtU5YnpWh6SLaFQVxjpW9HBtTPevvKUVCFjyJSu7i+VkUeSAKmfUafUgliAB4z9BTYFKNx1rjakpXRpui6masxgnHOa2UronREwUnipUjxS1vYVx/lnGeKYYz1702mJB5bAZpC2FxUSTSKGls0sR2tU0XcTVi7DyAc/WpCM8muhJolgBjmjbk1qpE2Ar3qvNbeYMdD1qnrGwGfPC0TVUuI2cNxj3rFU7vUrpqcr4g08ovmqCPUVz5Zs47V3UpJblXUkZmu6NFqto6Omcj0rxjxN4Yk0W5YFfkZjg4rSq7xuiVquU5e4QbyMVl31lnLAdKwja10Xa1rGJeWzDLY4qrFK0bYNbXbdkCdmWwQ465qORCcVndxlqehCo7DwcelN2ndn1obRir9RzDaAajb5uKL31FZtahsxSg7e2aL30DUerZ4pB9/AqXBotGpotib68SNVzk5r3DwPoQghjJXtV0oWldkVX7up6XplqI4wGHpWxEgVeK9WKscDZKKY5PfmqbIsMPPXk1BJnnmhMpMryN94ZGRWddPg9elS3YpPUxb6Yc8/wD165zV74IrDn8KwmzaGrOL1i9JJUHiudu2ZiTivKxFRs9CmtLlJUbfkdc1qWW8r8wrkqSVje90TOGB56UkkW5Mg81yhe5k3aOr+oFTaa+6UAnvWikaxeh0otC8O72rMuoGjYjmuSzi/IIy1KnkszcCp47YleaJSXNc2WgqQHJzUnkqKJSVtDRXGvBg8UnkZwTXOpO3mXcGg/OkRTGeRW8ZJ6MXUsI/ccU8yBscVnU5dgt1GnFVpY2LZqUrO40MyQaduyM10RkrFPyHoobnrViHCnmoqwvsF2TkA8ioW+U+1RTsnyktdx8co/ipzMCvSqkk0K2oRgqanUk1HtFsyZoQxnOTmpo0AGSKwcm3oQ9h5YdKBEW5NavQm9kNltzVRrbk5xSTbY4yDyEHanFFI7ZrWEGncrm0IWjQZJxVWZUX5hiqUOUlSkV5bgL3qtJccZzzWi3uyuUgFzk4NJNCsgz0zWrjZ3Q2rGZNZgvjFMNltPSuunUukjNrUeIjHirVtJg8mnNX1ZpTbsXB8wpJFBFcctylfchZ9pxkk02QM+KqL1u+oMkVdi8mozOQ+OtVH3nYVr6kyMJB705Yu9FS1N6Ar7E6W4ZcnrUbxBTnmsVJyGk7jkgBGfWpI4iD0NKo7LU1RYWMsOlCoQSDmslOKQutgEZLUrBl5oTTVmF9SxDcbAKfJesehoStuZyhfUgacnvULSc9amTsCiOW5KkAnFWEkH3s1Tu0ricRwuVZhipjIrLWdSF1Zi5WiuB8/SpDj1qIXsU2RTQKRmqUtud2ap3i7gpPqTRQl8c1p2dngZ71k6lnZ7s2lKyLaxlDTyoNZ2bOaTu9Buz5qrX1u5AweK2UuV2RF9RtqvlDcwp73yK1aSjdISjzS0FTUB2qRLoMetVNJxVgcbaCvIGOc04SZWpjZaES2Mu/i3kmsHUEC5raKaVkSuxTsH2XI+td5pN1mJfauhaK73/rzMqt2b9rPnBzitO2m3Ypxkjlmrl+CSrkc+K3jOxk0WYp/er1vJxVw1ZMttC2rH1p26tGtCRwb1qGZsDrSsLqZ9xIADWZcvknnFc1XVO5tG9yp5nz8VZhkzXJSbuaSVkWEkxUqy11J3MrEnmYpDKfWgaFE/8Atc04XBHek5K1gsOF0FXrzTlus96pT0DlZOkhIqUPtHJrSLkS+5R1HUhDGSrcj0rjNR1e5upSqFhk1y4ipyrlJ2Vy3o1pKWDytW2DsXrU01yocNWNY571GRk1Ml72pqOWPI6Uoi56VnKHM7sOpZtrVnPANdDpeklsGrp0nLUiTsdLp+mhQPlOBzxWxDagKOOgwa9iEUkcs2PMQAPAqpOg7d6UnYIlfZuOCKQW/Bx1rl5blX0sRvCV61XlG0elRJWKRAZNrVatp88Z6VVKetgkr7FtHDDNDDPNdW5FmQzgbTzWFqqcN8ufpXNWSSbNaW5yF/GdxJFZptmmkwB3r47EN+00PSTsjovD/h4Mwd0ru9J0pYkQBMV9LltCNOCstTz607s6C3twijA5xzU+0nmvWbscrHBCRRsJrKXmUtiJoMrSeTjoKxZSZIinAq3BH69aqAmWUTHbmpkiwRnBzWsVZ6ksk8nIBoEPOQKt2Yk9AeHIJ7Gq0sOTwOvWoq3typDTtuReTgn37VKIsDI5zWdOlJPUcpMnhUrjJzU6jk5rVu5KAjABznNJnrTatoFxUGRz+tKygdMUK4MrT24foOlU5rfHBNXShreRMmYGuWoeBlHGe/rXCXSmJyCOc10RVy0yNX4INc14x8OJq9kwI+YdDXVBKS5Q2dzxDWtNk068aKUEEdPes2Vd3A6VxTXLdF67ld9OV1ORWJqOmCMkqDxQny6r+vwIbk2U4JDHkNVpAH71rU946ac7MixkcUBjkAYNT1szRNbDnYHrUQYZ7miMdbgmxN5pQ2ee9NpdB7skHNKuc9KlvoNHoHw60MyuszjknINe5eG9PEUafLjgcV04ZN3bMK76HVW0IVeDV1OK9C10cjXcU9Oaidu1IizbIi5qF25OaCvQrTNgVl3suFbmpbRZzWrXnlxnLdBXG61qY5+b5q5ZtJHTShdnKXkxkYsTVVD5rbSMmvJrzTbR6EVoTpYtuBIrRitgiZGK45WtoVdFW7bacVUecrxms76Fw0IJ5A9RQBkm3L0FVG19TaCXU6jTL7zECsRmn3saHLHFYVrS0X9fmTblkUFePf2p7uo+6K5VG7Om1xm7NOCMxFOySKehI1ux70025CnNNSja9ieZEQRt3NPePK5IqOZX0KeupA2VNOVtwHrQ9dWaJaEgQtjmpFhGPmFXuS9NERyW2M+lR+TjmnomSNPy1JESRVqd1YpKxYjc4p2zIrBO0mKZDKu3vREC2KtyVheZcigOOtSrEymuRxXNdktjipJpORW8Y63IkLsJOasRMAMGtpJX1Mr3QsgzzVeUKMk0nFdBJ2M66kKnINQJdN3PFbqSRafQZPc4B5rPubxgDzVKzZpy21Mye/fccGiOcyHNaunypuxrpYlRS2CasCNtvJ5pc/Qm+pVnQo1EY3c4q4bcwWuTpbK/3hmoZIfKJwKrmTFsOjnx1NKZgepFZzimWn2G98055Nq01FbIT3K73BBxTBIWb61Ste5RMhbPy8e9WIZGU4JoqNPQdkXLdwTnOamlRZF4xXPpYLFf5lOMVZtzk81EmmtSkXYogelEsQU+9cvIZSbTItrZz0proepzV05LYpvqhjKRSbWIzV811qK+hEzEHmopXIOaXW7LQgBfoak8xlXaTTU+bQTSIjdeW1W4breBzWjs1cHG6LCyqw680oDMOO1ZJGO25IemDVZxk4obTQ473J7VcVrW4CpXPKF9R1HpoOZw3So881ppZGGxIp71FczgdcVMI3lcllKe7yNoxVRw2cnNdFRJMunoIrHPWpwxrG73ZpInWTIGSaf5o7tWiaeqMZIr3UikVgakc5xW0Fpcz2My1fZcAnnmux0afcoAbtXS43VjOfc6G2kOBzWpazEAZNZqRzzVzQiuDnrVqOcnvWl+5nZWLlvISea0rZq6qKT1MpaFxX96Gm65Nau1yExjXJ9aq3F13JqJSsWkUZrrjrWfPcZNefWqI2ihinnNWI2OKypaMT1JVfFOEnNbubTuhbknnU1pverclbULajfPx3pPtJHesZVEh2uwWcsetXLc7sGlCV5aBLYvKcLVG8u3HAzXbKXIjArGylux83Q01NCSJskLmsHT5pczQm76E/lLEuPSo3fmpmrGkVoMDFjxVu2s2k5NZwjzuxcnoWWstiio0ty74pV4ODsiU9De0vTwuCwrorG2CHpxXoUIWjqc8pXNmBdqireMCunQyVyCVyvpVcqXNc9R9C47D0tj6CntBgdM0cuonLUp3MeCSBWbcn3rnqu2xpG/UpOST1p0UhQ5zXOnrctq5dhuehJqX7QGHWu+EtLtmdn0IZZdw61Tng83Oe9TVXPoXFtGfPoIlySAc9zUEHhwRyZwB9K82rgYympW2N/a6HQabY7AMAccV0VnBjGBXp0VbQ5ajNKKJgOlS+XjqK2buzLoQzMU5zxUay8jPSs3a9iraE64IyDTXFYVCo6ixrzirtuvf1qqLTdhT8i0qY5wKkVSTxxWrXvWJJlTK05Yua2jo7kg0ee1QtFk0t3qNiLbkjnnmnm2XAzzWtluJvsCxYPApxBVs4rGcbvQLiH1PU01hz1qXuhgG9aduB7VCd3oNjguR6VBdwfLkeldVGPVmbMDVIC8ZLHOK8/1yARXLEjiqvbYuGqMktzTiRJHg9a6YS1HJaHn/wAQ/Bq3cP2qFfnXk4ryq8tTbOVPUHnis8TFt3GpLYrMc1Tu4VcGua97GjMa701iSRzVNDJA+HHeuqEkyIzJUU88UEAHNYt6nXHd2Gs2BjvQiZHBxT2Vx2bYwpzS7MelTzaCs0xwJAq7o9mb27SPGQT2pu1rhFnu3gXQxDbxkqOO9el6ZbhVBx2xXoUdI3OSra5swIAuTz71J0rpiY3b3EZj3qFzk0Nahcid8dagdj+FJsaKdw5weeKxtQnGCKhsaOL8QXZUHB5b3rhNWvCzEbq4MTUa2PRw+xl+eT1NSW8ypICa8eb5mdsUa32uF4xjANQSXbL0NYqPI3djUX1KE9wZCc1Xdj1JpSv0KfYgaYlsdq09OtvN+Y1nNOEW2arYvFPsxypx7VHPftIuM9Kzp35dRrXVlcE7sg1Mm5xS2TN1IljiY9QasLGQO9YOrdaoq5KrEDmp1jEi84qJyTWhElbUh+zkP04pJ4/l9KhW3YXuzOmUpkmo4nJYc1Sd9Wawd2aUaLsBNAbFbRu0RJgCWB61EYyc5rOcVzaEqViN4Sxp0UG3kk1sotaj5iQHa3NNmuNq00kKW5X80yHOangUgjNZ8+tiuhq2rDAB61OyA88VjOnd3MZOw3aByT1prhcV104XRg5Mha4CnmhZgT1okxq46S4wuM1Snu1PU80X93zHBMqPIG75qtK22qV1qzVIqzMxFZl0z88mtKTtLUvoZ8hO4nJq7YsDjNehN3REbs1Y4lODxT2UqOa4ZvWyNCnPliRjk0QwNuzTg+SOom7F2OE4zTXg3E7uaz50rjSKktmV6VTcNG/PWtqT5lZgpaimZvXpSq7SHHWtnZajYNCc5IqSOEelYpoE7ssJEc89KcyKprCUm37paZJAMVchjLnk1b0jbqDZKbcKeaZt2GuVb6ijIsQynIzVmQrtzUS9+9uhMt9CJiGHWm4D8Uua2qBdhGtyRkUse1V+YU4VOZcr3EyhdMPM46VCW3cVaj0NUtCaDAFPaLeOmM1Hw7CZRu4NmTUcEzA11Ql7tmJO5ZjuGDVdiu8LTkkKUbjvtBY4qRdpPNYTtaxMotIt28amrWDjFZaqRlKQ6MY70P061ahcyb1InfavLZqlPLk8muiktClqVlPz9anK7hWU5XTuithnlkHpUhwoGax1ehXNfYA4xwahkl2nOa1Sa0QupSubk5PzVk3sx5zXTBOxnLcyXuSj5z3rp/Dd6z456113tHX+vxM5K52NpLuUYrSgkK9TXNzanK+zLkM3fOatwz89aTm9iOW5oWkvTmtSGcAcmu7Dv3TKaJjdgDqBUMl6BnmtpdyVErPqOByapz6jno1cVSrpqaxhqVGuGkPegIWOa4pNzZpsTLEanSM1qlysh66EgQ9aNhHNW73sK6GMWHrUbM3c1zznLYrQiklx3qL7Rk4zWMqrWhSRbtstzWpbtgemK7sPb4mZT1LiPuFMMCE5PNdzt1MXfYkXaOmKZIwJPSk2kBUuDjmqUjknrXn1pG0dUW9Pt/NcE1vwwLGg6fh3rqwtNcvN1Mqr1siC55bHanWNtukDVnN81TQd7ROgtF8sDntWlbSYxnmu1OxzvU1LeVccnnGame4GODWkpKwkm3oRAbjUsUPOc81zavUb1J0Q4znrTJQQpPSt7kmVey9fX61kTvuJNclbfQ1jsQ9etMZwp5rBuyNIircY4qRbnJ5pqrYcokgmB/Gp4yCMitoTuyGmS/LihYfMbOKc2tkSjUs7IccVtW9t3IwK2pQdjObuXFjwBUEzbTnP/wBarq6ExKMr7jTU4NZwVy3cvQjP/wBenSR5/rWVXcFoNRTnoAKuQY+lKjKw3qWY+QKsRL+FdLsQSqOuRTxnFXtqSIQCckUbAe3FDTbuA/ywAKRlwK1E9xuMc01uKxk7IdiJj39ajL981hOT6FIbv5p+786Ud9Bj0kxwaSd8qea7IuyM2lcxL1SVYGuC8SxHzSaduhUHbQ51+/NMRua3pysU9h08CXULI4yCMGvJPH/ho2Ny82w+W56gdKutdwuKF76HBSxYJ2nIqvIpHUVwJ3eps2RhAeoqhqNkJQSBzimmYvTVGa7bTgGmmTIz6VvvqdiuN++OaVSUHWpk2tB3tcMliOacRgUm7MbsIpycZ613fw90UzSLKyj5unFXa7sT0ue5+G9OEUajH+eK66zj2BfbnNenTVkcU9WXV6cdKdWuyMyN2qCVvlpoZBI+DnvVaSXgnP5VLRSRn3dwRk5rndXvdqHnFZy0NEjhtfvt7EA/hXH3xZnJzXl4qS3PQoLQpMcDOahMx6Zrzo76HSmWIJmUDrU/nnuamr3ZoMY55pCu4dKjlVtAZA0WHzWtp0ypis5Xe5pG7RZuWDrkGqLrmos72Li2SwqDwau2sIJ7Vk7od9GaUNtxnFTfZQw5Fc1SNnexKkMmtCBwKijQocGsE+VWRqndWJWTcM1WcjOGq1C90QmU7uHcpIqgilZK3S5VY6IGlEGeMYpywNu5FUpK1upE9C0kACc1E8POBWUE+bUxUtRrII196bnI5rocW1cSkQysV5FU5nOcmqgla7LjqFvIAcn1q+sycGsbdy5J20JxcheQf1pTqSqPvVv7Oy3MdXoQtqmTnIpraqM8nNXBq9v6/MHSInvRIc5Gaj+2lTwaOVJscY62HG+3ryar3EoYZzWDT6GnLZlU3JXqaZJdYOSwreWkdg5RgmD1DPAG54p21QmU5LUZyTxToY9p9q6k9kJbXLsFyoYAsABWtbBJ0zxWM4NtjmnYlbTUcZC0semgHpzXM4cvuv8Ar8DLn01HS2vlqeBxVNl+aplBGlOVyKRl71nXaAk4xWkNNEaOOtyoY+3enINjYNd8UmS+xOpBGaNxBrGULXLVtxROc4zU24MMk1LhZaA0SwHnk9avQHaetZzWoX0LKAOeTQ1tluK56kXci7RIkJHpSXKEL15rNU3FaEqWpUAYnk1NECOc0STexq5E4kyMVFKvrUrR3EivLbBuhyagaAg9KuMrSdy0yVYsdKdv28Gn8QEdwglQ1nCExufSqckmKN7jt4BHFWoTux704O7dy5q2pZAxyacrZbFc801IhF+1IU9c1Ze4VeOtPVu5hOLbEE4IzkUxpsDGc1rzWehlYqzSP1zVOSUhuTWkJWvctIfEpfmpDLs4qIvm0CW9gabIyagmuOODmqlTtqhRZAt2QeTTZ5iy04d2XIzpHbdyQKrXgyuTXSm2ZzMecfNmtrw1cYbGa3ktLGLO7sJf3YrQjlIIrnmlc52rsuQTH6VailrKW5LXUu21xt6mrqXgHeu2hUSX9f5mbQj3wH8VV5tQz3pVa+lv6/MqMSu1wzHrQilj1rgk3JFXLUMI696nSPHJreEHuZuRMq1Zjj46VvGnfUltjigo2DFDhqSmxGgDelQS21ZOlu7FKRQuIX54qukTbsnrXlyhLmsaqWhftzt71dil969Oi7Kxm0WFnPaneeTXSm9jNIUSE96Utx1oTvowZVnfOarBctXFWWqNFsaOnyeUM1ofbQRjIrvpyUYbmUlqLF++f61q2sOwA8VnGF5cwpMtiXZ1NI+qCM/exirlLlM7Fiz1VpXGDkVuW5MoBNTCV1oSXYUHHHSrKJit0rCtYcCqg81Su5/fgVV0CWpiXk4Lk7qzLiUdQ3WuOrI3iuhWacgfeqJrkE81zyZpZjPPOeDmpEdjzmsnK6sN6almMMcH1q3Fvzzmtad0TdFyGNnxnvWjbW44zXRFX3M5s1rSHoce1acMfHWuum9NDB7kxU7ee1U7lTnmlWegolVocHJxTUiywxUU32KbbNGBMKOM06SMYycVNTQIsjVTU0S5PNZp2ZoXYhgZIqaMetdiXu6mLetyYYApw96fNZiDvzQDg0X1AcKbIcCtL6BYi3c8mms3Nc7kr6jsRSHqagcnrkVjOy2Lj5jQ+D1p6zDPWpUhEoPcGmSHcCQ31rqhJ2JaM2+HUn864fxGwLMO44q1K+44o5WYbWquze9dEd7lMlilOQCcCqfiHRYtYsHidQ2RxW6s1Zkxvc8P1vQpdKvpIZVPynrWNcqobivLqRcZ2Rra+pAkeabPDjIOK0WrJavqcqxz2/GkUDNdLujpT0HEe9KQMdayKSAkKaZv561SjfVhuW9Ns2ubhVGSSa9v8A6J5ESBh0HFXT96RE3aNz1PSbbYinAHFbEKkDJr04LQ4WycHikJya06iImY9zUMzADmlfXQbWuhTmkyTzVKefAzmpbsNGRqN3hT8w/xrkNc1EgEDisqrSRtT1djidSvWMjc5rKnmD9eteHiZKUrnq046FUgM2KabcZ6/jXLeyNLEsar9aH4Puah6stb2ZIi/LT1AFJO+iCQ2VVYUQHy2FKWjsyo3LokJX60GE7cmovyu7LTGxsQ3NaVlyQT3rOrJfeXJaaGzbJ8oNW0VR2FYON0rnO3qR3Cgg4qjIgDZx0rlcLVGbRbFQjFVbxMLkVsoXkmS3aRTLErzVZoiz5rVa6tHbDuaFiuOCK0PKVhnisqloyTMajEeLI4qNYjnmrWxzqW6GTxAr0qm7FTzW9O7WokyI5IJqrNFk5HWqktbItOxGq7DzSPcEDg0mtrmkZDGuXC9SaryXbDqa0TbdmUtxn25l4LVBNfHd1rSMdb2HfUauokDk80xtUO7k5raNNv+v8AgkXuyxBqIPU0971T0bis3SadimyGW4VxxUIm7HmhR3Q+g/LD5hUb3Td6cbS0XQTImnYjmhWZupNVZoa0QhV1Oea0dNvHTCk0k3JbhutToLW53gZNXEI9a55+ZzyiLMoMZz1NZk0eCeaxc47sKdyjNGck1TkQk4pwve7OlaoiKhW+alKq4yMCupPqJrqPSE+vNRyqQaLuUQiwjx3xUwAx1qWmhu/QegxznrVqKTJ61GrepSWhZErKMg09bvA55NZpsm1y3DKJBSSnHWk+xi1Z2IWUEZxT4SKjZsp3aJjGOtRyBT1NTyApMaqAHnFEkKsOKwaZopdRn2bFRSQ/N6VpTUt2Jy1EKjFV5YwOwocBq5AsQ39KsogUZ71cYvdlSloPD+tIgy3WiouZ6ExdtS9E4QClZ+c5qZRuZX1ANuOM05lwuc0K6RD3InyarTpg5pv3mCY1LgLwKa82WzmteblRbjcUPvFKUUDmlUmmkieVlS42ryKgScEkZzVQXRg2U7uU7s1VmuNynPauqEdLEyfumXcvk4zVzw/PtuAM9TXY9jDfQ7/Tp8IvNasUoyOa4ZvQx5S5DJViOYdSazaJaJorjknNSC4PrVRk07IVtdRTOSOtNDEnrWUm3uGxKgqxGAOSa0hDqRJlmNwBQ92ifxV1aJCUWxYbxXbqK0IphjOa1g+xE4tDzLTTNz1pNq4rDllp5ZWHWk7NOxNivKi9Tiq/kjORXNUpJ62LuxwhxzTxxQ4tbBe5Ir+9J53PWrc7agiRLjjrQ9wSOtUpaXZNivJLUYlGetc82+a5ZNHdY4zVu2lLnrVRm5ITRsWbKoHNXvtYRfvCuxPlRhy3epUutVVRw/PtWadRM0gG7kmuGrVtIOp0/h4KwDMenv1rqrdgirXbRso3MS1HNzxipWulUZzW+4FO61ELn5hzWTd6nuPWsJzsaRTuZNxeFjnNUZbkk8nFcU5NnQlYrvcc4zTBKWOB61zVJsuxat4GfrWhBaZHNaQp31IlPUvw23TvmrKW4Brq5bKxlct28YzWhbRHg/hVQWpDZqW8eME9xV2NSOT9K7oR5UZPckI+tQSpWdVMa2GfZw1ILYBsk1nHTYeq3JVGOKHNS3dagiPODViHr1IrOL94p6l2LgY7mpa7ubS6Muoqtk88U4MCevFQnqOw7NKDTs7XDYAwHJNRyMD1OfSk5txsNbld3xz0pomGeTWDkm7l2sIz5HWoJG75rN90BEznNIsh3f1oWoXZajfecZpZMDv061102+VNmTvcydSlKREjjtXCavOZZW74rbRIuKOeujk5qoWzWsSmhA3NWoZAy7SRXRBrZk6bnJeP/DK31q08afOOSRxXjF7bOkzKRyDjpXJiqbclJDjPQhVHianyqWTNZuyXmF9DjguaCADkdK1uddtRV5GajdsHpSXxFXsBYsADxSxqM81dtHYlHXeBdKNzeeYVOBg17v4V04RQx8c4HNbUd7oyqPodnZphQAO1aEa8DJ5r0bW2OVjiwHeo3ejUlasjZu5qGV8c0Mb3M+4faOTWVeXIGeazab1Q0c/qt6Ah5JwK4XX9RLMwycj3rmxD9w7MPH3tTmpGkmJOCartGwPIrw6j1seit7IWOEE5NSPBxXPJO2pbbRUd/LPWozLlupq4pPUu2hailUjlqGl9Dmm421IS11JIlZxk08QNkVzzmpGtkiX5k69amScbdpIovzItbaCxLvfNaEOEwa5ajuN6mla3HHJq0zblyOtO+iaMpKzIWmOcGoyS/SsaivqNDWUrUUilxzVU7xTG2typPAVGQOKZCozg9aIytodUHeOhbj+X2qdZQCOamSU3oZTJkkRh1oOB+NbadTne5FKPlrNuF+bNEXYIvUjBBqG5dVHYVvFOVgdyjPdKBwaotd/PgGtOXU1p67kqXAcVHOQfemoq+5tLQozSEVWeUA81000Q3oQNLzwc0oOeTXTHVGLHiRkHWnLdEjFKpC+pUNh0bljyaczEHOawaVzZPUtQzrjBpJIVPzDmsuVL3kKzuNEQ7ipVt8HcKU52V0UStH8vNVxmN80opfEgS1saunXpJGa3YZAyA5rlrN9P6/EJQEmfI4PFVZELcgVzXsgirEDpjg1QuEAckVrCWl0Uiq+HbpSpw3PauveIO+xKJRTJW30NWVhctisyNnvRl0FaKXRj5hy3TKRVy3lLck1FT8S/MuRvuA5qZQp61g9dDORYgIXvxSXV0i9SKxu+axm9yt9vjx1qa1mWVgAf1pzi9rmlnY0GG2PIqrJuzms3K1rmUbMbG2W57VZXZjmh2kN36CuARVaTAOTRFiTK7yAHpUUkisOlWlrqbWZXYkHI4FKJivWqh8LTFoxkt3hvapI5xjOc0409C2rRJ47jPfFK8x9aiV+YysEN4N2M9KtG5DAYNElaKT3JlHUNwxyaguGyppU1u2ZszHnKyYzxTpJunNaTi7XNEOW52jrzS/bN3BrKUHIq1xrjze9Q/ZwpODXRFSVjmqXTK9xBkVmXKFTiumm3dkMzp1ySafpkpiuB9a6oPSxklqd1pdyCgzzxWzbyg45rhrXT3/r7wehoRPgVOr+9S5a6GZIj89alDgc5pt6g9hRJk81YjORnNJR1FIkEgUcmhr1Y+prV6Ewg5MqS6tk4BzVK5v5Cc5Nc1Su73T/r7z0IUElqWNNvnDjOT61vQXwKDmt6eIcl/X+ZzYil1RN9rz3o+0c5JrV1Fc5LEqz+9SicHvWkZdibPcY8+aaJfeokNjxMPWjzAe/NG6sSMeX3qMze9YybvYoPtGOppPtOT1pub6BZjZJsiq7zEd646s3a9wuILshuuKv2uobBnNPD1rN3f9feRLUuLre04Bok1osvDCrqYxWtFktPqZ11qpOfmyaht9QYyDLfWvKniJOW/wDX3g1od34a1AeUue9dRFqA2539OetfQYeonBIwsLL4gihUndz9ayrzxavVWBqqtdR6lRg2Z0viYyHIcUxNZ3nlic1xPE8+psoDjc+YM5qCZgwNE5XWhSIAC7Vo2dmTjPesKcHKRcpaWNe3tcY/lVxIvavUhFJHO9SaPgcVMiFvrQ3qDLdvFg4IrVtYxtHWqiZyZowpjFTovIzXVHQyaHtwM1C5BNTV1KSDnrTSeen1rF6aDYoHGaY7EVnLQpIYp5yfwq3F2NKnHmYpFqIe+Klziu21lYzEUYpQSDyazb1uUxxk2ikWXIznJq+ayFYDJ6momc/nWU32LSIZX9TzURbHOeawlq9B+oFqawLVMUNtWGslN2c1UfiJbY9SetKzFhjBFdL+El7mZqnzQkZ4wa4TU1KSspIqoJpah5oxLvAz6VRc9wK6I3LQ0tTonIYHNbwdtyX2LMyrc25RgDuGCDXkfjXw5/ZeqGbblH/KtJxvC5K3OSuVV5SR0qvclVXANeXZuTNEcaoyKa3X6V0PRnVEYZdpIzTclqaSQO/UAvepraFpJFXrk07jPXvh7owjhjfaOeteu6NbeXGDjFdeHhZXZzVX0OgtkwoyeKsHgV3XOdoY596hZ8d6LgiNpOcZqGZ/epv0AzLybrzWBqN1tzUNWRpFHK6vfg7gW/KuO1W58xznvXn4mXutHfRWolgkZTmm3axZ4Arx5rXU6Yp3KR4JxxUUshAOcmsbu5va25m3U5+hqv5zf3q2jHQOhNDKx71o2duzYL9KmtJqNjWMTTjhWMDA4qUbcdBXmuT5inEjaPd05pI7Q554qpSafuvca1L1tAqDJ61KUHWs+V9RapksOUPWtCKUbeamUmiZK5DLINxqS3AYc0RatqQ1oTPbbhmq0luVPNdKje1jKUiKVQAc1DHGuTXPWottNG9KbsEuEPBpoJo2WhpK44F88ZqeLOeaqM7MznsTSbTHWResFY4NdM4KyZjBvqZz3RUnBqnc3LMOOa2jbVo3Su9TNuJHNVPMZT71tT940SSJoLkgc1OJfM4JpypKJTW7IbhSfbFZ8q8nvTpzfQzuR7GNPTj611Qb6Gb1Jeopg2g1pOzCF9ixEQKlEXmVxNa3ZvsPWDDVMowAD2rJp2aL6kqIGPIqwsQxmo5e4mBTIORVZ4Du55qPaKOhUSxaW21sitq1V9oGK5qs5WuU/MmIx1pQqtWcY33MnchuLcnJArPmtjnmnGFv6/4BSkVWtcMcU1rdl5xXRGXRlX1IjG2c4pHjYDNaRnqNoFprIGNHN77YrAYAe1PA8sdaTTvcWrJreU5qz5hI4qX1YNak8W9lPNZ+oLNyRmpbtYi9nYzwz5wSa0NOuWSQbs10tcyuO+ljprZhLGD2ps8Yrgqx11MYvWxXChDQWIrKMHuaJ3DeQO5qKSQv2xWm240rsrSow5quxz2q4y6M1SEzxSYz1pp3djN6EEsXzHFMjYocZrdPQq9yxE/pTZpjjrzWF3cVtRkG53rSjO0cnpVct3cib1JS4I4NVblsA80JWZi97GefvEmkdgBk9aq7uVEYWY03JVgSacXbc0UuxMlyo4zUhmBXNb09VqYTuVZ5wAazbtt5zVx00M0UZR6VHBxMPrW0bozszr9LmIiA9q27WbGPWuSrrIGjSinyBzVpJRjmpiupFiWNwTkmpPM461NnbUlpiq+DUyTAdTVwYNFa/wBT8gfSsltYaVjzisp1Gna/9fed+HorluMa9wwyc5rRsityuWrklrHVm0lbUthVh+71qaC5bdjJzRRdnuZVI80S+khxk0/zzXWpHmyjqPW4IHWnC755rWM3HW5DiH2jJzmke6296n22o+W4xtQIPWmrqJJPNVCr/X9MXJ1JPtW4Uvm+9N7iI3n9TTBcc9axqTa0EyObUUiXrVKTV0ZvvVjb3df6/EzUW9R0V8sh61biucj71ctSa5XyvX+vMrlZKHLd6JJWA61y801rckqyyMx5pIZD5grPXm1JW512h6iIEBL4Na02vkr8pzXr0qvJT0Yoq5nzahLK2Sx/Oq7Ssx61jUqN7mqXUFB65qaNyo604TaWorlyCc8DNWdxYV0SnzR0Fsx9smZBmt+wiBAyOavDR0CZpxRcU4pXc99DG5JHE3bvV22g/iPamo31BsuxRDgmrtsNprSO+hLLyPxUiyYGRzXQ+hFhHkNQ5y3Nc8m3IpbEykAUlK19WJA3TFJ5Rbk1MldjJI4RxweKnVMVpTgQ2She4pQuTW7VguOAIFISazldx0C43BPvSe9SrooN1Nbk9elK2lgZFIajP1xWErp3GrjST1Bp0R3daNQd7D9n40CI9fWtIw1uxN6WYvlYHoKbt5I9K6FG8bshFW7g3A+g7VxHiW1MMxYDAP8AOraXQqJyd0MkkmqL8VpErUjLCgNhutbJAWbeX15rO8WaIur6e6lVyB8p963jqrGctHc8O1a1l066eOTggkD3rObLnnnNedVXLojoT7nLRfOeKSVDVzstTdqxD5fU96XaRzTSvqx2YoUA5HetvwxYG7v4+OAwJq7rcGnue6eENM8iJPl4GAK7+whCKvPTiu6j8Kscc3eRppwKfnIrqWiMmiGV8A1VklwCTUgkV2uAOhqu9yPU1PmXy9TNvZxtPzdK5nVrkDd8361nOWjKprXU47V735iGPWsCcCVt2e9eRiJN6np000Is3ljAqMyFjzXDOSS1OiKa3I5JQDUMkoZTWLimky+Vsom2Nw/BpW0iQc8nNae15Fyg7li0sCj/ADcmtWMYGBXJUfM7s6Ya6skJIHNCuDWSjdsOo9H70/zQvJNEYa3KHR3IBxmpd7NhgaJ3SuJruSCUjr1qRLhsVldJNMSHq+49at2xxg1cOWSuZz7F0S7cZqC6kB5reFkjlaKEsoIPNVw/PWnNGsW0KxJANPj61yy0bRtzaFhFUU7dtGaUF1ZlPcguLny161h6hdZJ+Y12Jqwop7lASbj1pZEBGal6PQ2RA0W7jHXvVd9OLEkVpTk46svZEf2XZ1p8UeCK6OdO9y1qgnUsKqyQhRnvSg0nZGbiQMpXqKReK6VJWuieVC5yOarvuBzmq5rArIljZqt20hB5qKkW9UWpX0Litk8VNHDvPrXJUkyiylrz0qdYcdazcrg2O8gUw2/Oa5amxUGWLW3Ibp3rVgi2jkVlJqURz3JXtw4zVOQGKTnpWUZu5ENdC1DskTtUM9qh+tdj1XmYtyUis1l3xUUtrkY6CoUX1/r8DRSu9Sv9i68Uhstw9hSjJvQtzsVZrQo1RLFk962g+pXNdCSRMB6VGd3Q1vdyViovQfEhxnmrcR9eTWMpatFWVi3bydjVl7eOaPDCueMnLQynHqjPm0oBtyii3ssOCwxW8ZtRs/6/Ax3Zu2wQRhQae4yKhvnM7NPUqmAsScUjIVGcVnbWxalchk45NQs4NRU3RrEeV3Lk1UlGM4q3G/vDTZCFz1qGXcjcc1rHdWLeo2TOz1NVvmLY5rWPVEx6lqIkAZolx1NRJXWhLYQyBTmp2uwoxkVpC2xDTeo0Xy+tRvPv5zxTstUxNEDEFutNb5qyjroKVxPMCiopJM5rVq+gR0KzT+W3Wla93CuiMLRJm7srTXJHJP41Xe5LVskmtSHtcrtJnvU1mFeQbqdnujJ+R0tjhUArXt32qK5aurJuy7DNnrVqOfPBrBtrZja0LCyhRyactyK0vfVEJMR7wJzmoTqyk4zQmun9fiUo3dzI1bUmbIDVmQXzFgCTXM7XZ69CC5UjRSbeASau2l+0TAKeDXPHVNIVWJuWOZwGJrRjtgpziolBqS1OZyJZD5aZFVBOzPySK6VI5asOpaD/ACjmmGUg9aU562OdRE+0470x7gnqay9o7jsQtMcUwXG080OrroSxw1EA9acdUGetdUcRG2/9feZsb9u380y4uyEyDzXPUrRvf+vzIkzJuriRj1NQJvLZ5rmniG3oyYzuWYpGjGcmrtvenIyayjV1u/6/E03Row3oI60s10NvWm/6/q5HKQpIZD3qXYw5FZ1l1FJJEsV7NHxk1oWl+zYLN+tXTqXsv6/MSsXDegCmf2ii8kiuxRiyuhJFfBzwavQsHXJNTLyEi1BwR2q8nzdKqEr3QnuWLdCpzWzYTAAAnpXZh5W0YpamrE+RUu6vQcUZ8pPDz61cicKMjvQlZCaLEcnOSatRPn1AppdUDRaiJIHNPDmqctRbg0mRjPNKi5Pr71m730F5koHanDk4q7u9hrYcFB6mlQDJOacI6kkyKAAT1qTHB962StsSOHA5pwIHWmAUEDGTT5RDGO2mNz3FZNu90UN659qRz2rNxt1KvdkMp45qMMc81k9WXbQTHNSxLzxVtNom9ydFNP2bTmtqV7ambeoFWNQuOatvQVyOQDHNcp4rtd6FscLThdatBHVnB3sZBOay5eDWtOxpZkRwabnnNbJ3FZ9SWOTBzV2NhLGQRz71tTlZ2Id7XPNvin4V+X7bAuMdQOa8sBKvgmubERs2zWEr6nKxMU5NKZuTUtNs65PqNDc5NKWz1OKegJj4sMeBXefD3S98olZc5OPrVU0r6g9Ez23w/aeVFH7AV1FquAOK9KnHS5wy31LamlY461oRLQrzNke9UbmQjPWi+gIz5ZyCaqS3YXvWV+he5mXt3wRurl9Vu9wYZ/EVlN6G0Fqcdq0+4nnp3rKW7bdtrya76HqU7cuo8yZOSaUTK3FcLXM/Q2IZ0Zjx3qSC18wYPes5uy0GtdiVbDyG3VejeNkAI5rC/OtUaRXMhjxpuyuM0igjpzU8vQ1exHKSBmmRqx5zRf3rDTRPHwQCaWdTjIq5SS1Jv1K8bbXyTWnazIMZ61FRtq5ck2TShW+YUxQxrictbvYSJFJHXmrME2TitoSViJK5cVyw5qC5bg1tTqLYwcWjNnJ3daIDk9zVVJ3V0aWui0F46Uqx85rC6aJvYkB2iobibaOtaRjqQzOu593OayLhmY960d46lxVyvGrBuatINy80J31OiK0HrAOvpQwCijnTewW0KF24U+5qujtnODiuiHVstbCmTNMZST9avlIViNoRjk5qFlx0FaRvYhsYR6mmsu4c1tdWJV2KiEfWrcAxxiplIpIuxx+9WITtb2rlk/ItstJMOKlEqVjOPYQeZnmpohvNcs1dGiVi9bxqpyavoisMVlypO7M5N3JPL+Xis+5gZmPFQ0m9BU5WYyG3lU5FW47d265zW3tG0ohUcdWPktigJK81B5G5skVbd1Y54y6ohkt1A6VEyFeO1KKSeprzXK8tvvJNVfs20mlGVpcptB6EU8RqMRAL0rpVTRxKTECEHNWY4vlzWN020ypPTQB8jcVOLoqBk5pQdpag9UWIroOvNQu4J44oUUpttkONtR8NxsPJq5Fcb60tZaGU4t6krYIzUUjHGKzkrsyiivKBtquBhuayqRvobx2JAy9M1FJGpJNF2lYpb3IXjA5FVpNoPvWyTUboOZiEBhVeRvJOT0rWF+ol2D7UpGKikkyOtbOD6C2IvtBU8GkMu/k03psUtCPzCDUqTYGSaGrq5EthpnBz60JcYHNS4pbDtfQRpeeO9QySkZpwVnqZyKU0m49TmmK/Gcmu1fCZN2GSSbhVZmI55oj0M5Mjd8dTT7O52Sjmtl3J5jqtNl3oDmtRZcCuGrvdAlqTx3BA5qxHcc1yuS2K1JPtZ6EmnG7wv3qt7E2Ks90z8AmoSp27i1TzNM1hZMoXUhycmqqsN2SawbTd0erS7ly3uMsFya0rZQ7Ag1ldqVxV4nQaTK4ZVLfLn1roY3GwZYc+9YzfMzjaGkh+GPFQyRqpOKvVrUxqIha428E1E91mlzX3OflI/PNRyT+9RzJMaWozz80MxYHnNRVkrNoznEjSF5G4qwLBtu6uVVZ8t0clR6jREUOCaHTdzWdWpK5HTUgkgzTVhI7VmpkbDxb7qlW1IHShz1uWpO4KHQ96sxL5nWuv2uiuXzXWhpWtso5q0YVx0rZQU1qiHuMe1DdqjEZjPArNUXGV/wCvyAV97jHNR/Z5D61203yq6Dm0sWbWJwwzWzbB9oBJrldRqTaKjsaNtG3pWlDF0rppPTUUty2sYA5p8chR+tdUXyu5NzTtLohRz+dWluTnrmu3nuhIsw3J/GrMcxPeqUroJRLcEhI5q/C5wOatMllpJQPrQZTmrlYjqOQEmrMfI560473YSJAfejpyKJ6PmBDRJmpI2x1qIyGyXfjtT0fnmt1LYm2g4MSck8U/69aa12FYXdxigtn61XNoCRGzDNIeT1zWKbbsNiH5eaaT/wDXqaiHF9SF1IJJpu3vWHW43IXYaeg2nr3rZLoSidTkVIp4rfyJY4qcZz1qCRcEmk21sIglzg5wBWLrFv5sRHUHNbxlpYZ5/rFr9nlIx16Vh3C9c0oamj7FRsUwnFbRE0IG5zVq3mK9Sa1vrcTtYfq2nR6nZvEwDblrwTxdoEmkalJH0B+YU66vC9iab1Z5wvqaQk9hXO77s7mrjGZh1pBk9TTjLsC30L1hGZJFUDkmvZfAOlGGCPIBxg8VcI+9qwm7RPVNJhKKvpwa2Y1x1r0YtM4JEwO0UjvxVSZLKsz+9Z9zL79KlsaMm7mwTz07VlXV5jPNZ3RrGNzGv9Q2j71c7f3nmEheM1zVWdlOBSbTPtCHJ61h6jpr2spxmvDnV5pNHYtCi7ucLzVu2i+QE0nsap6BK4GR1p9rc7DyBWSjdWZcEXZbhSnUVRa5IbioUElcuCs2x6XGetTx3K9Cam1zR6hIyPyvNKg2rnFQoty0BJ21Gtkn0ppc9Cac5WGhpXuKliDDmsZVGap6F+1y3XNWXTYuaynDmWhDWpErZbHOPWrcAXPNZSptO9xNdi/CoYU6Sz3LkVnCVmYzMq7sGD5p9vaADpXROrpypCT0LH2fAzUZG2rgrohshmlCjJNZ1zdE11wjpdijG+pRecucd6jZN5xWb13Noxdh6WY6mnrAAazb5VY0i2SbQqkmqNyxyaUZWd0aJ6lNoi7cg05YAFrucuazE0QyQ8jFAgYnIq1Pl3CwNAcdKhaA56GhVG0Q11K88ZUelMjQua64aIzZOsBBye1SxFUPNY1ZWehcdWTrOrjAp4mKjgZrCd0rGnLfQQTv34pn2t1bqcVVtNxJItW99k/PmtG3uh1GKxnTUZXZbuXYLjzGC1rwRHYGFclWOpjU0Hlyo+aq7FZXwDVwpR5dDC+uhbhgC4Jq5HEu3IAzUey6kSlcjnQEfMKpzR4HAquWz8iYsptExfk0SQqBUO7Zs5diqy7c1XYZbJFFSDbTNovQr3DpUYORxV8ttWXBdwQDPNSZ7DpUN+8acrYjgCmCNpKuSbSBaasnSFo154qJyVPWlKN2Te5E0jZzVi2uTnLVanb3RyjoXopy5qTvyKGczViKVSTwKhaPGcnms2rNDhIZs755qvMXzx0qVpO7N4PuRqT3JqKSM5zXXZtGcpakLzeXVO5maQ04bq49tSvvOcZpzSDbzXVzA2R+YOtAkOOtRruKwjvj61E0zY604xJFilyeTUjHHNTOHvaDvYYZ9veonn8wHmtVDS5g027kLoSM1Vld0PFaQldWZm2Rea+aRpCVrWyRLK8xwMmoreX96PrW1PYzb1Ov0ibEIrUSfNcFa13/AF+pSJomZj3q1GQo5rkUS2xTJ3qCa5KdTU9bCWu5CLsHnNK11uGM1M7m1NFWf5hmq2GPSs4q2jPRpy01LNpDK7DqK6PS7B3A4NKSvsRXrRSN+HTXRNwHSnpNIh2tniuWTcZKL/U5IVOYmF1kgZqUv5i+lavblHJWKksZJJqs5KnmuaKd7MxktSNpCahd2NZSbuGlhY2Oasx5I6UqkbqyZhURNCRG+SKtG6UrgUoNJNHNOF9Sq3zNn1pQma5JxbWhDGmImnJDntWLdtzFomSDHQVLsAHSrTTeqGiJ1GadAoBzW8Wm9Bl+GXAFWo5Aa7oS92xViXIxSeTvOa2eotkSx2wHPWphboRwtaSpq2hBLBbYPYVoQwqAOK5VDU0WhdhXBrStwpHFdtCKe5E+5Nsz0qNo2GCQaqrugTLFuxFXk5Arppu8UwJ4jVuJiSOa2UbMG7l6B/U1dhlAXrWsexJYEvHLU9HBpyV9hFqNh3qbfhsVV7IkUSc5BpDJSlJuwlrqOj55NP3gVKXQNRyuPWn7snijmcdEPSw4SY707z+e1XzNKwt9WL52e9RvOV5zSlUWwJdBqzAyckfnUhkGf/r0oVLDaYjy5WmFzmidRXBLuN35604Dnn8qx3DYkVc0uOea6rJIzb1Hx9e9SYxyTT8wYoYkc1G/I61pawipNlfU1SuE3qfWnC97ikcR4mtgjEjnHNcddHGRk1ty2ZpzXVyow9ajYitb9A16BnnrT1dgR0pq1wL9rNuXaT9a434m+Gxc6e95GMtGpLHHatpfA0iVoz5sLDPTmkPBrjbex3oaU3nJpBFznmmlbRAlZG74TszdaggwSARmvefCNiIoVwMHGa3pJN6mdW9jurBAEArQQhRXdHsjkaHbuM1E74HNN6E2Ks7g9+tZd9IFU81Mm0NGFf3GAeee9c9qF+VzzWDa6nTBXOevr/LH5jmst7omQcmuGvUex3042dzStL9Au0kVT1WSOUE5rw5xbndHW42MGSNd3Bp6tge1Wou25KuROctQvSlZtXNIppEclwUOCTToXEnrWqTa1BJk6xEnvUotiwOM1y1JWlZbFJaXJIoyh+bIq0rLjvQk2m0aN6aFeT5mOCacluzjNZyViovS7HG32n5qnhgJNczelh8xajjMZqQsWGDzTUujFe+omMDjJpUdg1ZrrcF5mhZSFiK0413LSjTehhVdireRg9qrhMc4rT2dzNPQQn3qtO+M1104aolvUzbqYqTnNZk0xY47VtyuzRrEmtLUPgsc1cNqinOK4KvNexshWVQOBUAjBbrVQSlowjoMuE+XANVPKy3NVUstDWL0B7cdab5YoipR2D0GeQGOSKkW2zWl3e7DqEtqOuKpTpg4FbU3ZaiepSnhJb1pbeEBuc11Qm2Zy1WhbZIwO9U5gA2QcVMk2gjdbiwt82KuJHuXrWLuy9iKYbTUIGTnNaX00KSuPORg1LFdFDyTxRNSkVFdDTsbvc4JOK34NSCRjnNc0k1v/X4mVWLYsl6sueaqpKUmyD3q/h1RzpWZrRz5jBJ5pE1IwsctwKltPVbEct9CRtTSQdefaonuNxqJ6ISh3DaCMmq7sA3JrFppDIpApUnNUZW25rWUVozWBnznJwD3pq5XqaU730OqOxIp9elO7jBqJqxS7kyx5GTU1vtjPIzTptuVpbGbd1ZDriRTnFZzsS/Qmqb10FBdx6L83NSeUuQRWSX2imy3bELVvKsK10aOWa1DaMcmqc/3iBSqRdk0ECMA96Ci455rOMOZ6mtytIm05qF2A6muinfYm9ypON3NUph6VvCK6AmUpCyHk1DNcH1rpik1cbYxbg9zUomAHBNU431BN9BDLv70xmNS1YlCocHNLLKR3NJ6Fbld5C3rTFlIOScVSbM3KzsTiQMnWoX29SKmMXczkQsq5zSMqgZJ5rVRkQUbuQdKpRSkSj612UY+7qcs3aR1Gl3J8sZzW7ZFpCMjFcWIjubJmg0ywjnFJ9qD9DXJGDjqyrX1F873qjfs+0lTWdnF2Kv1KcDSLkk4zUitIH5JNOq7u5UHcsplxitHTtGa4YEjrXK3Z6nS58qOjsvDZBGEJzXQab4emjYNswKqnFqX9f5Hn1KvM9Tq7fQt1tuZBnGfwrlPENmLSclR19KrE01yKfUrCz1sZULFj3q4jdq402zstcCN56VDNbkGlOF9UZzWpWeIg8ComjrkqJ2M7ixxZOasKoXFSk9DOeqJMAjNKse6s5w1sjBjvJOacEIqGnAxQ9Yt3Wpo7asnRk9RNE/k7VqCVTmivCVlYghMZJpyIQOaUU4SvcLkyA1YjYrXRTm+W7YyzFJkc08z4PBrspzutR2JIJSx9quxgk133vZomxdt4s1oQwZXJqFTE2yQIQ2BV20U8Amtaa96wnsX0VVXmn7BIKuVnoTqkNNuY+amhfB5pwvHRjTbLUZzyKsRkiurUosRuc1ajlYJmrUrCJUnbqec1PHOfWhPUTLMVx6mpPOy3U076isTB/enb/eqZI4SEdzSmXB61DeuhSHLLnrUgkAPrQ3rYTiOMpPNNDkd6G+wLYQykck1HJOTWc3ZFIiEp55+tOW4bqTiojdD3HfaSR1P50vn7u5Jqm9BCCcg9aljnIPrUp6hJdSZJiTjPFS7/euvnTiZWQgmx3P1qQEmiF9mD0Q4E9zUbtnpWji3pEV0ROCRkiqs6kA54rSnHqQ9TjvFn+qbjufxrgboYdu3Na31uXHVFR8+tMPNalaiHOc0obmmtg0epPbS7H9zU+r2q6lpNzCRnzImX65BreHvaMylpqfHuOf60YPrXJszuixfurnNOjO8iiTd7ou53fw90vfMJSOv+Ne2aBbbETHTHSt8O7q7Mq0lsdRbDCg4q0G7V3pnI9BGbFQSyc09wXcqTy5X7wrJvpgMnNZspLsczql2EDHI/OuT1S/xnnk+9ctaaOulEwridnJ+bmqss2wZJ5ryqtTsehTWhGt+UP3jQ94ZOprklOz/AK/zOm1yvIWJyKfFl+OTWbkpxsiUiwlgzLupktuyCiNRBcozowfpmrVrDgZ71OtjSPkXBgDmpYHAbk1nI0asrExCvzUbRt2pp8mhn1COPDcg1pQovl4xWTjzR1Kb7DZIg3IFOhjwc1zSXsxJ6FuOPeOlMmtSoyBzWEV9pjvYhRSGwalVAfr3rZNOWoXL9kgQ5rTRwABXTHXVmFTVjJ4gwzVOQAVSSuZFWQHrmq8i7s9apPsU2UL1VQGsto8vmrbcVdmsC7bqVXNPMgPOea45X3ibpABnkt1p4hJ5FTzWeotdxjWx781E8BXjFVfnaGpXQLFxzUElvlvlroa+0NTsCw+tSj5Bz0FROd9h3GSSK4NUpYwecVUZWiLlbIHiUDpzSRxjrjrXVGSeonF7kVzuTpmqhRn5OaFJtlK249IWFXIWKrWzacbEyZHPlzSRwDGTWDdti4y0CVNo9qrkZPWtIztJFJluzdkIzwK0vtB2/e4pVXa7BongZpACDxTmd42yTmuOVV7EOKTsWItQbGKjnmYjINEpWRm4q5US9dJOTxWvZ3ySKMnk1alzJX/r8SqlK6ui01wuODVV5Tkk0pLU5UmIXBHJqne/dytQ22bU99TId2V+c1NEwYc9acWnE7UlYlPAoDfnTkk2K9y0jkrSM3PNZt9jK1mMZsdTTAw3ZpwXu2FZknDHg1JHHzyaUb7ImTaRME2jmgTkNgGrS5dzK1xzSsG5NMZt3PrTc7oCIk56mmSuVHXk0RXKh3uVpHJ5JqtJJmqjKzLexSubggnrWfPdOK6aeruRZFd7ncMd6iA3NnNbp2d0EtBGUDvTd2D1rSm21Yz5mh6TD1pTMCaiUO5WoqMWNEoytR6hJlcnA5qKXrmtlYzvrccjnb1pryHPFO/QRGz45zzTGlz3rSN7EMp3PJJqtFG8koAHJNdNPY45/EdnoWlny1LVuxQiAV5Neo3I6IorXoaToaZCWjHWs3UaVjUlWQnvUoQy9BmsJSAgltfKOTRCnmt0qZXasOL6o0rOyywyO9dVpMCKVGBmoUb+pM5NnY6JDExGccmu30rSbeZRlQa76Ci4+Z59WVi/d6dHb2rFea8m8aXCwXJJPtisMXZQOjBu8jmre+UvgHkmt6ztXuEDYNcHsZJXPTqPk3JXtmgbkUyRQw96Le7cxk7q6KsiAdqgkhzzXFUTexityFY3Q09Qc80ugpyTJl5FSx9OKnS5lIlAz1pdlE4JmLVhycVZiYVpGNtUJoWRwBzUDfNzXPXi5bGViMinpGTXFWTWgrMmjjxUyQkjNXSbXulkqRkVIttu5Irtp3egN2JUj2GrUBOa9Kk3azF0Nayj3DmtBF+Wt7kN2HcYqxbuQapy5XdEonZyec1JDcbTzk1MKmty0rosrKJRQdg6GujSSTMtmW7Yj61eSPPNdEVdXKuSJFg81Mg29qpIaZJjH1p6Ng5PegGSq5FTK5JxRy9QRYjbIzmpgQRwad7kdQ+hp20mjlG3oPA4pwUk0RV9xXFJ4pAai3QdriMMgmmCE5yaUh3toL5BJzimGI5pOLJuBjPY1EVYH2qZRaKTuORjnkVKjc9alK2rHboWEfC881IHyMZFaRS6kPQAcHNTRcHJ5zWsLCZZ2blzTDHg57V0rTRGV7jGAUk96zdRl2ISTyRXRGPUlbnB+I7kSMRu57c1xt2Dv69amL11NYlVutMPuau+pQ3PpSBjVrYQ5G2kHPNaVnMWXae4xVwdnqTJXPjs4U8nJpjN2BrHc7biKWb3xV2zi3OBilPcGr6Hrfw/07ZDGcYJPP6V6rpMREYI47V14f4dTCs9bG5CpwBUhOK6ranOiGSTA96qyyepp3KKNxN1OTWNqVwQp5H5VlJlpnH6zfcuQeB+tclqF2zsTnNefiGlqz0KMVcz3nOOTVWWR2PFeVPWR3RtYYfU9akQZ4rKTsXrcuQWfm9e9WYLAI2TWF2r2JcruxcbCLx0qlMd5rJJ2KSKzwbjyaeq7F44roukrGkRpyx5NPjzniolYssRt6mpo25yalPmQnEnGGHAqWKNgMis78iaJfujlzux2qeNAazcVbUVyzFtTqakdgw4xWKScXcHqQNBuOaZtMfJpJXWwKw+KcjoavQzE4ya6+bRJGc0id5TsJqjM7M1WtTJETNgZNQF8mk3bYpakFxGJF6is6WII3Wi6d0zSFxYjvG2ntCwX5c1hK8Xyo3bshIYZGIBzV5YSq9aibuyXJdBjIT703ydx6U00tBOQjW+O1R+Tkc1sk3oZptkbW+DmoXUnihPU1UrjFi555pssXGTVxj3LUtSlOCDmo1ZutbQjeNina9hGXcKRYVJpKTT0CSBo9p4ppJUYHFaQdyUtCEylSSaT7QfWptq5M0UUg3F+vSnxx7m6VbfvA12JjCcegqVImYDmsp1JbId7IvWu6LqasuokTOea5Zpt2E5LcrbGRuTVqIoy4aqWqszOavsUL1VVzjpS2lwO5wK0iuiL1cS6lwOu7iiS53DihzSWpnyX1KzXLg46il3NKO+DUc13oXypDJLPjdimrCFNQk1dl81yTyjinR2+WyauN9GxcxOyYXA61XfKvg1O0rAtRrISaRYmJ6mq57MHoSCP9KlSXb1rVKxhLVgbonjNNMoBznmlKSHy2QhlB5Jp4mGMk042ZnNEU1wAagNyp6mqaurAloRSygjg1Wkb3pKGpRRuuec1nysG4NdUI2Qr6FOdgDxUInKn71dcI20IZItxz1pkkpJ60lGzuS2kMDntT1Yk5NVNXHdWJkk29TU4k3Lk4qHEmW1ypMeevWoXfjrRFCIzI2OtNMhHU1ry6k36EbSZo31pa2hHkRvE0pAFbWhaGWcSOKU5qKsYWVzq40WBAAMfSmySg968ySbdzoWxWkmAqBphnGaiS0uUhUkywyc1p6fhn61lfoDeg7UlLEAClsYNvWomtNDOL0NW2QZBrY09zuFTexbXc67RVcspBNdrpN1LAPm5GK1pVJX5kcdRJuxY1HWm8pk9RXlHjeKSRmcKSM1pXqKUTbCR5Z2MHw3pz3d3znANenafpqwWoyvOKwhNTWptjJvnsZGrqqscGsonIIrn5bt22RpG/KQtGc5NHl5rGojOb1A2/OSKikjKmuecZJmd7kTN2JqSOXA61kpEtFiGUHqanBBraGqZlJMTHNSK2OtVFa2JtcikJJ60+FCaT2EyTyc1IseK4ZQbd2iGTIgqxHH8tb0qNveYhwTmrMUfHWumEUrsbHtGAM0kLhW5NbXa0QWubNgwZetXlGRXVFtLUyejHolSoQvJNJrqHUd5uT1qaLDDitKcFJF20HsCvINNEzdyc1M24S0M3qaFlL71sQ5I6130btakFhRnrS4IFa2Y7huNO3VLVmUSxtgc1Kp79aWtx3JEm21Ik5NJuxLRYjYMRVhTmr3QmOOG5p2BimtEIcY93NIEH41L8ht2HpDkZPcdakWLjJOfrVqNtyW76oRoxjHFR+VznNKdmtAiL5P4elMNsD82KLbFXGG25yRxTXhKnIGKcqd9mCmKqmngccGseXULkkYOOalTI5JrRdyGTRyZ75ApzuQMmuqBLRTvr5Yl6gYrldZ1cuzgNxjjnrVTqW0Q1E5PUZS5LMwx6Vz91wacGWmtio3Bz1qNj61qNoZnFISc1SECk5zmrdpMVlXHqK1ixNHyAST3oVSTgmsFPTY7Xa92W4YwcE1qaTbCa8jQdyKlp3BHtng6zCQJxjpXe6fEdoPpXdQ0RyyNSPIHvSsciuq6sY9SCbiqNw2KlsFfqZ11JgHtWDqs4VSSTx1rGb0NI32Rxet3G7d82K5e4myTzmvMrxurM9KirIpyOC2KEwW5NebUujsW1iUxg8imDAbHesWuVczHFtvU0LWQqBzVozkjg1lUkr7Dsr3GPcMFwarbiz80N8sXIqMdSdU3DpUcyY5pLVXK2ZGhQnmpMBeQc1MlJGl2xqsxfircKsOSalS1sU9iyi4HNWYWOMGoqO7uZy1Q90UDpTA5StYRbMm7bird4OGNTxzZGSRROmlsCdkPE4XvUM12DkdKhpcuoXIYpiX68Vp2sgJ5NOlZoJouMQVqlOQCa35bPQ54vUqyyDB9arGXBrOo0tjWIySXevFVJIGbms01c2i7EkEPl4JNWQwIxWMvelcbdw3hDntUizK56itOVXuJIXAz1FNcbORWck20JMcAHGTUcg29K1imtxPchc7uOlQMNuT3q7LohoiQHfTpgNuMZpSk3ojXqUZk56VGIs8YqvabFpaXHfZuOlRm2YNkGqU3zXQ1JDvKY9cmmyWpKkjNbRaTFKRn3MRXOarDOehpxlfQqLLdvGSeRWjBABg4oqvXQGWWtQVzRHCFNcbd3ZGadyZYS3an+S4HfFZuo+Yd7EUqHHI5qJXKnvVuaeg4akE6M/NRwKVbnrThUcS9y4sW4jmpGj8vvSm23Ym/QZtUcnmmm4UHAqKcXzXRVrjxKX69KRxjkGqd76gt7EkXTJq3EikZqZ1G9jOV9xZI+OBVLyyZOamU9fdHGXVlkIoTJAqE8P7VpBa6kt3Y4jIqPZkGtrOxnfQgYEMajZjnk1lFtvY1T0I5JtvA6037QcYrXVOxNyCeV8feNQLK2clqvnewadCXzhio2kznmqvdaGbWpTuXyCKzZ2wetdVMnqUZpCT1qAv6iu1a6oylLUA5pTJkdaGhN6AJOetL5jDnNU+zFcekvqalExxWMtCrkUjH1qNmGM5pxvYmTGMeM5qJ3Oetax1M23cjJ560+NGkbGTVN9SubQ3NI0jewZhke9dJCiW6AAAV59WbmyFuMlucd+arTXBFJlpakDTBuc1GZBnrWM9WWh8TktWpZy7BShH3glsW8mXnJqzbR8jFZztzELQ1ba2LYwK2dNsHLDINcsuZyE5HX6TbmBAWWt61uM/Lmu+PLCGpyPXUlmtvNGSDXOeJtMSSFsrXLU2aNsPJqaOd8NWnkXJA6ZrtJ2MdqeucVjBtRvcuo+aqchqEkkkxJJ5qvk1nGb1R6HKrCeYMnNOVhnmlNXsc9WBZUKVycVUudozg0qsPdOSLdzPlb5jg1F5pB615nPaWxq+5Zt5iD1q5HNkV0Rl1MZIeGJ708HjNOMmr2JsIAS2TViMgDmtLXREl2Ea6CVJFdJJxWkaUepPKy1EhY8VcRMCqhGzsSw2HrUsecVaihEjIWWolt3L5wa09m9ENSRrWC+WBWpCpcZrp5TKROI8UyU7RWU01sSnqVjPtfmrtrODjmtKE0zV7FouGXrmoJHFLEpX5kZta3LFhcfPiuitCGjUn/8AXXThamhBdWlxuFde+qKTAoOuaADnNQ3d6gPTg8/lTwfSkloO4AnpUkZx70mncGyxDJ+Y71YSU4p81gv3HiXnFOWb3xUueomiZJc05fm6E1XN1QiwmAODijJJ4rRyvqQ7AqFuTTvL7nNTBNtibHBOcipQmB9K3jCyuwbZHLDk8cCovIqW9bBca9sWBweRSbMYA6jrUTp6gu4Khznn8akYEoT6dapQdgbuNQlW5pZJMr1ORTg9bAzD1VJGBwSPXPeuT1FyjncTkGq5Whowrx+WJ5BrIuiSTya0Rp6lRsjmozzW99LgxhNN3UlLUSQueakgba4Oe9XGTEfJkcDMc44p5j2n+tZNq9kdUtWSxvtFdF4MtPtN+JP7p6VUZdWVHqe5eGrYpEgxjgV1toMAc9q7qexxzbbLinHekeTAreOxDWpXmkwMk9aoXEw60mMyb65xz0A4rmdXvF5yRz71jPRGkI3ON1m8Viw9K564cs2RkV5lZnp0lyrUgZyDSq+TnNcE7Wubw1LkTZGKc8YzkCuepqi0nuODGMc04XGKmVr6leY8SbhzQBg0KKa1CL11LIcKoOearXN2BkcVpCKCWruZ0lyS2QatQSsUBOcmnU5UrLc2hoieF/m5FWvOrjaSbNLEkcxIGSatRSgd80+ToZTJHuMrxyaiMrE1tC0UYSXcgmDfeGajivmQ7WPelB3ujTRotpc+YOKc0Rfms5QI1QRRbW57VaimMZzmlTTjuNtSJG1HA96ia88wd60c76f1+Znya3K88jYzUCsW9eaj0LS0JFh5HWrq2ylOlSou4OWhXlh2tikjTn1qZR12By0EnUnioowyk0NKOo4vSxYhUselWWgJGSKm9watqV5GMYqsbobsEVslpYlaj8hh0qCU80uWxpDexEvBzT2G4Vhf3rGzaImty3NC22eop8umoc2lxTCBwaaYQTxWsVoDYhwvWkYgrjFaRtcmxSubcNngVUNptIxUua5rI0h5liGLAq3GAuDSctGW10LActgVPDEpOTWPXQzkrLQmZApzSEgc1Kikyd9yu56kiqrpls0TtdWNIoayUCPjOKptWRSFyynjrS72f7xraNmiWRzsdvWoIUbzOR1qZ2i7oqD0NB1VIwah8zcadk7tkLuSI20c1Ik+DjJxXK9JaFrUuRvuTNNKqTz1okrPzMGmmMkz0qBsLW6a2EvIA4HWonlwx5ro5kC3I9+Bmq08uGzioi/euyyIybutM6HOapuzJYjgEcmq8q46URetyUR7+KTeMc9a1UbbMbKlzKQCRyaoSPvJzXTTimrmDRVkXNNZMLXQmkkQROpzTOlaRkiXsC9akPSnLULjehxTi5/Koaux30EZs1A0mOAOKqAr9BfNXHQ1Ex71oo2ZEwjiklOAv41tabpTAhmWsqs1FWIT0OhtgsEdEtxgZzXHa7dyuhTkucHrVd5zuyTQ9i0NMueaFJJrP1K6GhaWrOuSOauQwsrYrGMve0JZqW0O7AI61safp3IJHHvWVrtkylobtpaoo6AmtWz2x4NHJyvmMG29zXi1OJQAR0FaelXEUkoINXOreSiRytI32RHjwODWbqVmJoWUjmoqWvYE2jBtNIWC4LgDBOav3nERGO1YNLltFG0NZXOXvIsyH+lVXhx1FcySTsevTV0QmLHPpVeV2jOeaTbT0HVp3iRPqZQdaqzanvzzXRUmnD+v8zzeSxVa8yeTQs++vMqQ1uaOGhagY1cifHXrVU/eSOeW5ZSTNSqwNau17ozloP3Z7U9ckVfNoQyC4gZhxSWcMgfnNNNpXFc2rbIHNXIyMV0ppmbXYdkDrUkLKW61rRs3qhNaF1EVhmnBADXXypPUzaJocbq1LX7uQc1MrpEvUsE4qG4GQcVjKSeiEjGvJDE5Jp1pqHOCa4YVrVHE6Ero0ob/AHDGam8zeK6a07ohruSWhKSfWuisZvlXJ7Vvg7cpk7GjDJu5PNWAQB1r0rroLcjaTmhXzWTetyrWJY8k8c1KEPfvVJB1HBcGnAcelKwEkdP3YNOw0KCTUi9axe5XQmjOBUiyECtEyPQkExPepo3J/wDr0+bQnl6kwcYyacr7s1rFq1yWiRSOKeOR2rWT0sShGGTTSOMVnZMYmOOaaU5zVv3tBIQLlsGnpHg5NVKyVgWmgxoR1zVaTg9en61PL1BMpXcfmZHGT3rj9ctyhZsZq1G5UNzlrvuD0zWXcHnIpxRbKp96Y3TitXsJkbGmknFNJD6CZOeuBTlODWiSQj5ZztGKYXOfasHZy1Om6FQb2x613/w808Ahz1JH5VSSukNrQ9k0aLy0GewxW/bHgV6EFpc45In8zA5pkkvHWtVYlK+pVnlBHXNZl5chBUspIwNUvdqk7uK43WdSOSd4rkrzsrHXRijk9Ru3kY1nGZyPWvNlNPQ77JRsMMjLyepo849zWT5bGlJWNbTNrqM8mtFoQBnFclSKuau/Ui8rIqI221s1zzRT7DxGQKY3X1rWLuK2tiNjJngVEUds7utXzJtIrQZ9mAOe9XIIBtFY1HrdFxJCmDwKlEfAOaykrO/cty0J4kBFOJ29Dird7aGTeuo5JgOD+dSqVfnih6IzkibapXHBqlc2fO4CpSaITC2jZW5q+Mqvrmtrp6MJdg+b0qNnIPrUq1ggPEe4c9acLfb0rNW1E5MjkgP8XemBAp4rRx00GncmhHIzV1BxzUp63RLIp1B7VCiD1pNiVxZAMVFHHuapaRce5bULEM4oN0OlRTp3eoNXRWnw2eayLo+XJntmuiMbDje9izbXYcYI5pJTk8D8amatFtFxjyysRc7u+aswwM+CTmueKV7msnZEwh25Jpkjp6810KGqZlzcz0IeCfWh4gRms9mzS7RA8RY4ApPszIKIS6MtyVrETxknkU17fcOKmpJJ3NE+o1YSp71ZSMlKV76scmhysEIBqxEdxGKuUVZSRnLuWNh289aYUxx1NYSvfQi402+5TzzVWWEoaa+IuErOzIT1zT0G41q1GRo0RXEZU5pIwWHvQnZGbuI0RY+tSRoF5PWplaTK8h0jblxmoBE2SR0pSlZXQJdxu5g2DmrEeCM5FRZuzBuxbibC4Bp6KMk1o6d5XZhKYyUgdOaqO4Vua19nZ+pKkxrTLzVaSbnmteVLVjQhuMCoJWyc9amaV1YpELNimFytU1rYbfQFmyMGo5z7Vfs9bJkXsyq8mOlReZVxVtAbIpG9TVSXAJ710QvsjJlZ2O7imlq3SWxLQ0nNRsKaRAxVIPNOO8mtNA0YYoORU6B6iEgHJNQyk5yKtIzmxFV25xT7e3eeQACqukR5s6LStHEahnFapVUXAWvPqNSYyCaRx0Bqu5Y9s0k1vc0itCGRGxzULA0SknsWKq+tWLdN0gz0rKbJkb9oqrGKnRAXrmcVe6BbmlYw5IOK3rYbEGadLfUzmXoW4zmrCzHb1pVZMgegkk6ZNaOlmeGUEZ4rBO6u9wkdVa37bBuPOKnWfzT81Tdt2YnHQjniVctmsnULkAEdAKuTsmmXS3MC5lXcSSKpTXIzxXC2tGe7h4XIDPk4NQXHzL1rKpP3bdTedPQyLtTu7iq5Q4yaJOVkjy6tOzG7c1Ys4NzcmnGPO7HPKVkzTEIRRR0NJR5XZGHMTRt71YjOe9bqKtdkMlB9aekozjNLlVyGiwhyOgqaJBuyKpRUtzNom3YNTROa0TtsCCV2qawSRnDEcZrqw6fNoRJ6GwgCJk9aZ5oNds4psyXmSRNhhnvWvZgOOvNEo30Qmy00XGahl4rllHlbZNzndamCbj6daxI9VVXwGFeRWmoVr7HXBXRo2uqgHqM1s2V8JQPmzWkK8ZXVxVEats+SDWnBNtUc/rXo4Z8uxgaEF4Bj5hkep61cF2GXAr0VUUlYTVg35qRH9ah2uNPQtRsAo9anU7gK3T0I2H7ckUqrUpD2JET1FOCk8mk9x3DbkZqWPgjFTyjb0sTKueacsZJosK9kP8o5609QR3NPl1sK44Mcc05Jdp6nP6UXadkOxKJiOc8GpUfPfNac1ybDwckHNOVOM1Mpa6EtdhpQ5xT9meK6KWqEwaPPPejHHNU4u12K5G1VZBk5zwah36jRXnT5fpXM6/GqxEkdqqF7AlZnC3QySSMVlXBBYntWqXYqz6lZuuaiYnPWqjYu9hjdaY2T3p9bksbn3pQTkGrEfL7oCM5FQMuDmsnq7HVeyuhIFYSj3Net+ALXFvExHXFaQjzNahJvlPTtOXaoJ54rWikPbivRW2pyvckMvqahkn9KaJsU7m4AXmse/vMBucfjWcti47nKazfnBAbNcfqNyZCdx59q4MRJbnoUUrGVMpJJz1qu+F4rzb3bujpW9yKSPeeDUkdtuIDDrUyfuplo17GFYehFXC273rjqzdzZajG3A8jFKI2YZNSle9xXVxUh5OaVoBjPenyO1i2+pFt5HGaJYflJAptLmRL8ijJvUnIxUsFwVIzVSSb0NobFoMXOami5OTWEo66l8pZjKjqKbOqnkHmnBS6HLJtSINjZznNSIxHGetaPTRg3ctxNxyakJUjmpcVLVmb30IioDZp7Tqo60RexTVxRMDUbyc5FVKHukRi7ipMR0q3G+V9amPK3YclZDJMntUUYy2SKVVvoEdS3DGM5q1sGKzjF7iluV51BziqkmUPB/ClKMuhUBN+4VJCpznOaqVrFaIsMAV5qjOCDkGhx1uTFkTyYU5OTWfPl2xitHLmVy7aklvbkHOSBU5HFUoXi7kud2I68ZpVuWhFZypItSUlZkE1+w61Va8Zm9K0lL3LDglzFu13SirG1l4bj2rmc9zRuzsTxwFgDiklgHelGLehlzakQtQ2SKja2KGlKEkldG0amtgMQxkinxQ5NCikxyehHLbHdUkCbeOtTzO1il7yLxwI8mqpkyx4pKzZnFEsZJX0qvcAVqlFaMV9SnIhzToEYNz0rGclE6FZosmEPQLNR0pXTRjK6ZG0O2omi5PNEny2sVF6ixwFzUv2cpU2aLlJbFeeJcHpmq8JIbArpjDoiL6O5OJyh5p4uxnk10pJqxg0MluRiqk9x6VLvswUSBpsc1CZjnJNaPXccUNknx0NQvd44J6VSgrIdhouQ3Uio3mA74q4wuEiL7Rh85okusj71HLZksgaQHNRGUg5FVCOuol5kM0p/GoHkzxmuiOi0JbWxFkE0FQw4qtSL6kZGKT3q7EsjJINKTxTaCyEyfWms+OtUldEq1xFQyNxWlZ6U0uGIrOrU5VYqFPmZfGkxqvIFTWelpG27FcMazbaN6lD3f6/zNAHaMVZtrbzzWdeajBtHLUjyRLb6SoQkjNZktn5ch4rz6VdyerM6UubcpXUag9aq+TvOAa76bbR0JaGhaaQZFyRVhdI8s5xRPm/r/hjPmdy1BEy9aswqQ3SsVfqNm3p6YQHvWir4IFWrbmT3LkByOuaniUs+KxqSd2I6HRrATYJ7108Gip5fAzxV04q3NJGMndjRo77uMgU4WU0XXpU1KTScug1K+gydW8vNcxrMrpk8jNccpa3ZtS0ZzM16QxBOTUDXGeaxqLXRn0WG2uKj7qbNnHXisGdTepUlTJ6ZqH7Pu5HWhXbsctWn1B7QirVpbnNXHmWiPJrR0uWpUPSnxWxfmtIpt2ZwvRCtC0XNCSkYpX6MNGrkwbcKdGrFq0bJehcj4FTox4q4x8yGTINxyasRAA9a0S1JepZSFW9DVqI+UuAB612Unyu5k1fccZC3emjjkmt4zu9ReRKkuCOa07K5APWr59dCZRL/ANqyKhmk3Aj1rnnKMtBWOd121eVDjNcRfO9tKcEivEx1J3U0zrovoLb6oUIJNb+ka3lwuf1rkpKNN6F1IXVzstLuxKg5zxmtRZDtzXs0qkrHIkI92ydG6VPZ6kdw3NVwrtTSZo4qxsQ3O9M5zVmJ8816MtbMxsXIiWwfWrkaEjIroXQl7Eq4FSJzUv4rC8yQLzmlMeT1q7CvqL5fy9f0pUGOtZa3KTLEQJGc1ICO9aKxNrjyRt54ppNE9EOw0dakWPJzUR94dxWUjmgSEU+aysNIljlwc1ZSUHk1MXzPUTXUlUBzmn7cV107GbY1hzkU1sAVcpt6CIjg1C6g8dqzTbYFedQFJ/SuY8RqAj5P+eK1jovME9dTgb8AMTWTcN1qlfZmm5WY+tRM1NDIyfemUAxM4FG73p3JsfLRn656U03CjrRKNzt8ifTyJ7uNV7mvavBlttt4x6AVdONnuTPRHdWhwozxj3q6JcDg13x1OR76DWn25FQTXPrQ+4WKF3dkZ5rndU1DCtis5PS5rBanGaxqJJbJ4+tc/NeGRua8uu79T0qasRyTAKc9azpZzv6k5rlgtdepskixbRvLyBVlA8bcilUa2DS9ixHMR1q9Z3IyN1ckk5ysbW0L5MUi5GDUYlCk55pOnyPUzs+pG84JJFMWQufatHaxrFaDnj/CkVscHmstWrIa1QyVEcEmoFiG7IqUnFXKgmWol2ip0UHvWTlI1WpLsz2pjQMTwTThUcTmnoxws3I6mnC1dDyKt1FJ6ojmVtSVQy0yaUgZBqkiL3Y2KUydeKZNEVOQamTcZJI1uk7Aku3j0pPO3HmrnUTVxWJ4MNyTVpG2jFZpJahNa2HB1A603ILcVo4pozV73J42IpxuCF61HLZag0QSXHOc/hUDO0pockky0iRbZuvSpEBXqayjdvUUmOZ+OtVJ3HXNbzXu6Ex3KpO/jvTlgGcnms1G5q2TJt6fpUgtgRmqbsrIzlowa2AHJqvcW4UdM0txRbuUZUGeaiNvk8DrUTWmh0RRtaNZgY3VoXWnqcMKwqX0aJk/eIVj8rg1HOAa3pO8bkN6iIuBUcgGewomCYxQrnGcmplt2XkcCpm0tTW9tyOVQmcio42+bpUxSaZpFlkfMlQG2fPQ1Dja7QOS6kiRleucVHPGD0NKo3a5K3IFtmPJFPWBvTmsX717m/MiWOJjgHNSNbt1FaL3o3MpTVyKWLbyRmo0i39qUk5SJ5tBceW3PamzT7hjpWlrILp6mfckjJNVUkIfOa2hO1rjHSsW5poLNW7TT0FoSOfl5qhLKS9OPdkpDC+Op4qGaUAZyatLUaZSa6YHk1C8xatrO90aWQzzXDU4ylupq/Mzl5ETOSck00zFevFXa+hDeliPzyT1pQ5xxTsiWRys1V2Y5ya0SVrozW40yY9aI5tvJq4oBkkm49KBIKOVid7ASGprHHWi/QlrQhJznnimlia0T0sZt6mppFtvcbq6GNBGo44rzcXK7seph4dSWNDKc1MsRQZNcPNZnRO1rDRE0jVq2EXl4zWNed48qZ42MlbQvs4K4rK1GPaCa46MWpHHTkYM6GSU1YgtQuD1r2oRaR2Js07WTyxg1YMobpUra5VuosQ+bJFXre0DkEmseUmWhoRR+WKmjYk1TdtyPM0LboPU1owR1EnfYzkzb0a4EDgtnriu60mVX2gng1vRmmrSMakep0FvYRTAEKDkdaZd6ECCyCuqpG8LIzjJJmTcaIcE4PSuY8QeHmdWO057GvMnRSurbnRGep5xrOlS2shYggZrOWUpwc8V51RWukfRYSfNFEsMvfvTm3May9orXPQ6jZBxikiz0xUN+9dGc4+6TeVu5PSpVHliuykl8Z5tVXIJ7oq3tT4NVC8EZqIyXtNf6/E8urTuy19oE65FQyDac5rRx5mYrTQdHIRVhJadknqKROs3Tmnm42mtIvWwWEGo7GxmrEWpD1roUL2l/X5kuJpWV6svetEEMuRXU43joY21I5pNg4qKJpZ2wAaiknewaLU0IbGUx7jmpYg0TYNXWpSpsjmuWllYY61OoJXJNc0E5MWzM/VXAiIJ5rzzX2zKcdK5MY7QszoomSGYjOamtb17eZWyeteSrp6HS1dHoHhjVRKqqW7YrsIWMsYJ9K9ej70LpnBLRkc6kDmqvmmN/WsZNuexqmbGm3u7CnP51u2zhsZOa9qi24XZgzQhcA5P5Vcil+WuqOxnbqOXOKlQ+9G8rj6FhDkAE9amUALWqtYkMcc00Kd2azluJbkycHqakHvyTQ9tCtSRUyKQRkgk81MrybsO4gjIOTUqnoDVQtYW+orknrUJ65rBpc1kMXLdMU+ORgeT0qpJJ2QItQT9jmrAcEA5HNb0pW0IkuwE8VE3zGtJaCGOOevWo3HI5qbvcRXlJxjNcl4ml2IR2xW62uEV7xwt63zHk8cVj3HU1SZpYrtkDmonOaqw0MJxTCfWqtpcH5DeD70A0nuB8lSXLDPzGoxOSe9aqz2NW+ht+E4zNqKdcA5r3bwumy3TtwDVRl71rFSl7up1UTgAEVIZwveumLMWiN7nvVK5vNuecUmyVF3Me+1DAOD171y2r6ngsN1YVZ9DsowOXvbjzGOTWXLu3YArzaj1OxIRoncZzSfZcjnrXM56lRety3a/uhtAqWRye1TOSb1LktRY1zUnMZ4JqGrMpSHLdyJTluWk60nJPctrqP3nHfNPSVl96U0uUuPmSeeW60/f8vPWsWuhbXUrPLhqkiJJ6Yp1VZWuCdy5HFuGDUwg244wKwT6kuVh7nYOetELhmHrRZvQylrqaEToF5AzTJCo54xWkUupi1cryEMCRVZxuOacp20KihEB60kkuOtU/eKsNTDA01xhuO1ZPR2Nookt32t8xq00oIyDzVcyaIqRuyIyknBqzDzyTTUjOSsiUyqB1qvNMT06U5PTUSXch2yOau2tvgAtWDkrWLk9NC20ixr0qnNcLk1pBK+pna7uVJrkjqcVRmu+eTW19TRIIrkFutW0nDD61k9NCuUsWyBm5NXwqqtJasyne4wruPSop4eKtNXsQtDMngw2cd6mtYVYcjmsHLudd7o1rOLy8HHFaAYOuDXPK5mUrhByRzVR05rShdEsjaQJxmoJmLdDWz8yooLaFmfPYVrbR5YrK/Nox1HqildJuJ5qCJPmxihK2hSehft41wN1WvKjPpWrjGxjNu5DcRKFOBWZKuGrB0242RdOT6kqPlecVZto1frWapN2ubSemhM8SDpTQgPHpWsYKGiMNepG8IYEYpiWxj5pcrTuJvSxFNACCaoSQkNk81UYtPUuLKd7GxHGTVNIz3qrbWNug/nGCKaZdo4rqir7mK1ZC9ySDzUCkOxJNJfFZleaGzfIvHNUJWdmxmtUuoJdxsdq0mSaV4AvYUk25WQETKB1qCVto4FaQ3sBHGDKeTSvB71s5WlYmTK7xlT0oVyOtaKzepkI8mTUTciqtbYCN0J6U3yyOSatSM+YYzYpm7rk1SfUSYCbHFI8nvRykyZD5uTjNTRLucZqpqyBRvKx0OlRbQOlaY9K8bEO8z2qa5YpFqBBjJ607b5rYBrFy0uE5W1L9vZ7E3GrETAZz1rzajbZ8/iZ8zuKzkNWdqtxlDjrWsIpyViKSMFRKz5wavRLJtya9lySijsQec6Pg9Ku28obqai146GiWljUtYhJjFacMXlisUmmc8272JS2RSx8HrUO1xrY0rJ8mtizwxFS9mZSOh0uzEuOM811OnWLxIDn3rWlSVuZGLfRnTaPuBAZsV0AjV0AYD/CvQpyXLZnPPRkElhG+eOTWRqejBwRiuapBXuXGT6nA+KPCqyRsVQHNeZatpElnclCO9eZi6XKuY97AVuhDHb45xipvJAryJxaVj2VJsY8Y61GxVTU01ZXYSQyS7CjGahF8uOWr0Kd+U5akLIgnuVYHDc1VE5Vs5NYPWVzyq66MvW94Vwc1YW8Ltg10UZ3dpf1+JyKKLSMCvJFIrkt1rScLO6JXcsxAk81My5Sml0YpPqinJExbipI7eQdzW0ZtKwOVja0Wwk3ZOea6OKycJkjFenCnJw5jmctWNNizPgitLTrAROGI+tGGXvXaIqbGkwRV5xWXKimUkHOa6MbK8dDOFyeCINgmppMKvXJrgpwaVzTyOf1242ox9K871i9DSEe9efi4Jux1Uypbzb+KW5GwZFeXUp8rujbqbHhfV/KcKX5B6V6jomoieJecnrXoUJJaJHHWjaRoyjzBxVOW36mira90JSsrBZu0MnNdNps+9FycZr08O/dIlqzUhk4FXImJGCa64u6JLSEbetPTJPWmnrqJlmJDkHAqfaRmt0tCL6iYNOVRms5X3HfQeo5qQHJ96m3VDuPDHpUiqWI604JgTLFkdaa0YB4/Wt4x93QnqRyL0xTVjyTkVxyT5jR2sSBAo5pPLJJ9K3cEzNPqOCFTnPFSISPyo5OXYEx/nY75pm/LdalN7MHYcAGHPNRTLt5zXTZJXJvqU7lvlIzzXI+J8BWOcn1q3rHQFucLet3PWsuY89aa7mt+5XfPU1ExzVoBjUxjSuO3QbmmluaYdT5Ddye9IhJrdqxXMuh1vgOBzeqeo3DNe4aGu2BewxThrLQ1m0bqShV61FLd7ec1qvMxS1KdxqHHJrMu9TBBHNTLRGigzA1HU+CAxH41zt5ctI2MkiuSrM7acUjOmUlsikXHfvXmVG27dTZWZOIl25qrcuFJwDUuKl8hbBaSDdzViZlC8VzzjLnsbS2TGwOSQM1OwPUGtZ6PYQ0YOc1JEuKzlFPUtdyYoMdaeiA1Cloy99RZItvNROx24qN7Mu+hDglxV2CMjk1dTqIsrLsX0ppumJwTWcLGY/zN65PWm7yh4PNW31E2W4JCwBJyakkbI61N30M+tyNctx2pfKBo0vqV6CrARTXs9wz3rJS1aJuVniaI9KFU96nW9jpjsL5eD0pyAn1rVJoiTGyht1TROwXrTj7rCVmhd5zzmn4+XcTmnutSJD4ZFB6iriNleKzS1JkmtyC5kI6mqTtnvVuOuglsQSfPVO4tyKvlad0WtCspKtg1bhkPrRO7RqtVcu29xsbGea1LZvOGaycZbpmdRaXLDLiq0xPI9aqWhlHuU3iZvWprWJtwOK5qz7Gqkjbt0+SkkBXNVyXVzK6TK0j9zVO4mA6VpGGxVrlKVyx60wOQ2DzTqWRukX7SQY6CrSOCOTRRte5lPcZKgbmmeWBVzppMlSEywNWI3O3NYe9zFvYV5AwweaqPHueqavqStBxiULkU+D5aNIuxSfcmB3HrS7QB061z3k2DVkNAIepJNqpkmuqnG+5nIzpZgGPOapXEmOSKO5rFFaR9454FRbVHPFZ2cloXayI5IzjpVWZCtdEXayJsiBiF61Wlf5vl4rSK5nzFxQz7R2NINhOTVPV2ZThcUXCpnFRSS7+1WklqiGrMrufmpksW8elLVMTjpchWNkbrStmtU7sL6EbZc01kC1otdEZNIj8sNzUMi4bANXG5k3ZjGBpHchatJsjSxUlfBzUYlz9a3UG0T5iBueaaz5PNVy6ktpsbk5zV2yyzDdU1ErF0372p0dkrFAFHWtm0093AYjmvDxElBts9SVZRiTvamMciiCPa+TXM6q5dDmqYhONi61xhMVX88h+tctr6nlzWo+W5wpqjKwmOCa0ow94ulGwJbBRwOtSiAkV2LTc7OhHLaEc4pI0KmkpvdAmjV0+Ur1NaX2jK5rbmUo6GU1qKjlqmUkAVzyvcDRsCCev51t2K/ODVrRamUjtvC8Hmyj867cWCwQAgH8a9PD006dzinL3rEMN+IJduR1roLC9W4wobJPvXNSn7zuOUdLs0JMRoTkfWqMrrMeoPvWtSFnqSlpczr+wS4RlIHNeeeL/AAsmGcLnvnFY16EalJqx2YWryzPO72A2spV+tVXmAWvmaj5ZcrPq6HvRRBLdACqUt3zyazjTeiubtFO4uyeh5qoblt3JzXZBu1jmqKwnn7j1phn5o5TycQi9bXAYcnmrG/Bzms4pxmmjiWisWIrkkVcgkB612/FLViaLsJ3DpU2GIolF3uibKwixN1NSxqdwAq6a9+xDtudX4dhJZNwrq44ICvP619PRhH2Z58209CF1gibt+HNQzXkcY4NZS5Y6hrIzptRZiRu4psUxY5z+deVVquTNeXoW0uAopstzlSSa05ikrHLeJ70JG5zyK841G5aSYnvXk4mfv2R0Q2FsZWBGeKuXRLR5riqSvJGjKNpcvBcBs969L8I6xvRQzdetb05NSt0ZhiNdTt7e4DoDuBz71NxItdFaPUw9CIxBWzWlYuRgZIrehUtoG5sW0x74rQt5N3eu+nLqQtNy3G471PG3zZzWt9biZdhbK59asjpXVGzVjMcEzR5WDnNHIrhcAtKBz6VM1ZFJkgHy8U5XwetZ8zQ9CzC+4U9lyK1T0sR1GBOMmmMCD7d/es5qMVdlIRfmOKkWPjk1dNCe5KIx70eVwfetZR6kpkbREVE6Ecis1G7uAK3qcD+dJK5I6/lWmiVrCKFydqbicVx3ieX5GyelUvIcbXOIvWIOTWY5yapLqavUiY1FT21BDTzUTmoDcjJpueau72QPQ+RACx5qVE4HGa6nE0SO58ARfvMnrxXsGmNtiUZPFENGaVIuxdkuMDrxWfd3wHerciYxdzJutUA/irJutT3Ofm/WsKtWy/r/ADOmFN3Mi8vRk85zWXNeqWwDzXnSqOVzqjDQTe74OaswWbSMCASK5XK+q3HZWL72oSPJzWdPbqzc0Q3uybN6EQtgnSmtEWPc0lfnubK9kh8SmMZxViLMtE/MahqWPsBI3ZOKa6hKx5ugWd7Ar571LGpU55qb6mqVtGTgbxSC2D1lNNO6H0E+xYfOKtrGFj5xQ23uTJ3Kcr4bgmkzmnB33DlJIwT608xkDkVbkleJLQ+IuD1q0iFsc80lK+xE0KylT96iMnOayT7iWxajUNjrV2G3Vk6D/GnJJkNWKGoWxU8CoEtiV96iT5XY2pvSw2SLbSxgYya0g77g/IR1UnJpFXHTOKcrboF2HMhPfmm9Bgk1Kva4uoqx5ORU8bFOpOaIyd7jlqrEdyxYcVVzjOeK03dyFtYj8xexpsrgitXF3CzKLRbnz3q1HBhc80nLoWuw+KCRn4rZtAY1AOTWKlZ8rCp2LW3fzULp83NVNcyMU9SaOGPbzyafFCA2RXM6Seoa9S9HwvPFV7mTBIBrfZWJSuzOuGcg4qi3mM3NZe05nojoSVhRFnrwakWyDcnpSk29UPmaJVjEXSmmYq1VG6ZD1Wovms4zmmtclTya3vchLWwCfPc1ahdWXrWcrPbc0a0GyuEPWoDOGbrQ3aNuo1HqTRoepNTAgDiueV2O9xhYg5FDzZHWojumVIgkvfL61Vm1N2yK7YyXclU76lYTMzZzmmXU/wApyOahO+jNOXUofazn5qesoxnNKNuhvKFkBuA3U1DNIK0m7aGajdlSdkNVGILEmtqcUtmPlaI2UE/epGUD+KtVC8biu0xoIzzTmUetZrmsT1IyoJ5prjb3pqLe5NyEsqnJpksgY1ooMTV9SFzt5zULS571vGKMnqM8z60zfls961s9jOV9xGINIY94OaNtETcq3EGwZ5qofvZrem7mbethC+Dmk3bzjtWzRm11LVpatcvha3rDQmUgt0rhxNbk0Fza2OksbNY1XIrVjZEXC8mvDxu2hcpSdhsitKehqB4pQcBTXDBpaMVrjo7GeTkg1Yj0iQnJFE68VsVypllNB3pyaqy6H5UtKGI1v/X5FQRILAAdKDbKOtae2ZshsluNhJqk0exjVuWmg4x6EsLbT1q/Dlu/WuqhNW1M6isW4UII6VaSEk1bcXqZsu20ezFa9gxBBzipbsRI7fwteiGVWJH+cV2z6vHJDsDgkivVo1IxpW7nHOD5rmLeykMWo03xA9pOMk7a8+pNU583Q25eaJ0T6+s8Yw/65pLfVF3fNJnPbNdbmnLmM1G6sW2u4iMhs8Vh688UsDdCcVakkrodNPmR5H4thWOVmXHFcfNcsD1NfO14x9tc+twcrwVypLcYOSTVaSfLEnNc/JaWh3t6XKk0+PaqzTYPWumETjqCpMw9aUOeuapo8qu7ss20uTjmr8YZscZrNL3kjj5WmW44HAyc1JHKUbBq/hloPkvsaNtdAD3qb7cM4BrshKOn9fqL2LLUUu/HOc1o2cAkYHIz710QpqU00YVINI6OwRokBBxT7rVZrcErJ+telJOEDkSTZlT65Kx5Y/nTU1Bn6sTXm1J9UaqIfaju61at7rPeuNVG3/X+YcpZWcnvTZ59qE5zXUp6CaOP8UXW4H5ua42bDyE15eKfvLuXFsfbJiQHFajR+ZF15rjk7qxonpcyLqB0k4z1roPC+oSQSAN7VcaqglIfJ7RHoWn6rlBu/nW5Z36uvLCu5YinL3W9TKWHkkXEKPg569KuwkKOvNXBK97nPKLW5dtphnk1pwSg454r0IO9kiWkXY5PxqzCd3Nbrcl7GhCQAKsIzfhW8XZkNEokHapQdwrS+oNBjBzRtBNDTYhwFG0FumaylroivMmiBUcGpd3FW1pqT1Gs5HfrUbsSfasKrvE0sEbHdU6A5zWmHT5TNkgyKd71unrYQxiMZzmmOA3viq2bC5CylCeMVBKSOvSlrYTMzUbnZEc56Z4rhdfu/MkIVs1adtC4JHK3j88ms+Q81cWaWI2NRN0pWEhhz60wjJ60MHoRN14phPzU46D1PkjHNTxOAwB5rd3elxxfY9D8Cw4AbGAefrXpNrPshHPNTCXvaM6JajLm+IU5YVj3mog5Gacqmly4w6Ix7u+JUgHFY9xfleprz61TSx1UkjPmu2mJAJpkcTdTXGp3ubStsWEO0gE1v6dNEsIz1oltcJR93QjvrwE4Q8VTCs/NLZFKPLG4xgRwaIwC2D0ob0uVZvUs+UrL2qS2hUNWEnLlsikrF8kBKoT4YnFZSuFtSJBg84FXYmVlqJtuLNd1csxRhiKna18sZ4pRUmrMxcrOzKzyHPTNQyTvj2NEvd0LSTZC7bu1NBINTGqkXy6F61QsoJpZiQeKpavmM+pAZSvNPivtvU5NXBtjcE0WPPDLkdaWGbLc0PR6mShoXo5OM5qdbgqMZqG+xNu4yeUP1qCN/mNJJyTkxoSaNmGQM1H5Rxz1qo3QXQhiOOadGmODmk027MVxJMqetQ5LVUVZaFX6j1kKHBFSLMp600lcXLcZK4weazbi42k/NWyjqQim12ScA80CZmHGSaptmzSW5YslZmywzWvHCCAMYrnlK8mZzLMUaLjgVOkYzWTWtxdCwqDbTJYs8mtI3sZp6jY12mpPMIok7Ivd6h9pbuc0wne3NE53jZCSsyUQqy9Oaqy2w3dBWKXLqJN7DRbgnmplRVHOKtXUinLQrXLBeSaqM26qdk7lR1QjOyDjiq8s5Y8VUZxtqXGCvcfFKdvNWI5Sp61m5a8yLaSJWzKOoqu8Txv7Vi5vdCTtoXIpsJg9akRgevWtLNu5DXYSUY5FVJZ9hOamK7gtdCtI285PeoXj5JqZ1LM3poi3bTUc0gI5qqcm3qaNK5SfGcg00sc9a1i1FltXVhpG05zUEtxvOBxWqvMlFSWU5I3ZpgbNap2QpMQyAc1E02TycVtBvlsZPuHmDseaRHZz1qdVqKL11JSoA5qvLJhqItttEyZXkBc8VGUYda1TaM+YMFlqvKCK0juTJkPIPPel6c1q2RuIc5oWQjmkiCO4feKosp54rakmjOVgS2kk521JDpszSY2mtZ1YrQxclex1Xh3w/KSGKE11sGhSlQCuK+Zx+LXPuXGndl6HQWOMir8GghWBIrxq+Ncup0ciL0WlRKOQKSTS4s9BXC8Q2w5dBqWUadBmh4Qo4GKfO29SrCRnbxVa6Ubsk1rFvQHvoUyDUcoOM4rrTVkWVpXfbg5qm/BzXZaySGnqEbDeK0rbgAk1vR10M6iL8AGOTVqNwp5rptZGDLcMmOlaFpLg9awlcVrmjHrv2IZyQRUtn45MlxsLnnjOawqYuVNJI6KeG5k5HTWGtLdR/Mc1LMFYblPWt3P2qOVx5WQfaZ4QQHbFPt9UuA/LH2rnU5r3blWjY0Y9RuZByTUF7PcMhyxrt5ppaiio3OJ8RWc0jFipK461x91ZsCSeBXnSipTs9z2sNO0dDPubJlG8c1l3LMhP61cqaTPQhU5kUJJCTmmLy2a2SsjGrKyuTIC1WI7ZpOxrOTsea1zMt29oYz1rTtMRtkj86xUnzG8cPdF3zk21WkI3EirnJKRpDD2FR2x1qWOUH7xxVQk+bQqVFdCyl75HIJNaul6n5jDdxXq4efvGFfC+5zHRQ64kcWCc49aoXetRzEjcK9Gu7+6zxnhmndFB7tCc7hUkN0McGvJrx5VoHs2tywlwDzmrMU49a4YtrUTiWY7jA61Dd3bMpCk1t7TTUm2py2s28kmSOlc+9nKj/dNebVm5SbYnYmit3DD5TWtbW7FMkVgyk1YbNYbz05q7ptgIzkAVwVarUbHfhI3NuFjH82avWt7Kvc1wQrOMryZ6boxa1NW11coRuJJ7elbNrqIdM7q+hweJU9DxsXh+XU0ba65BB5rWtboEY3V7dGSWtzzGu5ehnHTOav28hOD711xld3Ia0NGFwfrVtHz3rpWisiCROT6VOrFRzirXcGPQbhmlMZBzWluxDY4A4zTwoJz61Fle5V3YfgUuKctECE2ZNBiyevFZ8qtqguSJCByKkVMcnmuiELbEMDx1pCCfpSnoxobtpCAOanXcHsNlYBfes65fbxnOK2iScv4gviiHBx2rib+dmPJpdbmqta5j3Tc1Uc01GyNGyE89aiei7QiNuOSaaWpvUQwg1Gw9TS5tSkz5HZsU+BiZVPXmuxK6Bbo9M8GAxxqzd+a7RrsLH96s46bf1+J1NXasZd9fHB+bArDutRbk7sYrCrOyOiC0M6XUC5PPSs65mZ25rhk9WzWw61jzye9XG4XvWV1poV1IVVmfJyKtxyPGMZolLXQ0ewobc2TzWjaKDHk1lKWupctURXpVOQcCqsL7m46UPYULtFg7gMino7jmonUNI2JWu22470IC3JFVa+o9NWK0IPOMVLDCR34rnd76lc2hetz5ZFWJ5cx4yCTS5tmzBq+pUWF5GySMU97ZcdBWVR8zsWmr6AtkGHTFQyWyK3FZuPvWX9fgUnrYniUqvBqC4UnOTV05b6D2dyq5IHSoCrF6umteYrzNW0ti8XzGnNb7D1qZzd2iE9SeGRse1WFbcOaShdasynHUbMuRVQzMjcV0x0lZkpIsR3G4VLvUjmm0uhA8KpWmGPvScAv3InTLUqxgHkUlFIJu4SxAg4qnMpQ5B6UWSd0OD6MqTTSHrmqU2W61rF6luyINm01PBg9aaV3qJu5qWTx4rSiZSBWVSnqZtMmWLdzVhY9orGSaWgORIuc02YnFOBPUgUtu5qyseRWc4tuxcmMKYOKkEK4z3reNMhyYhcKcE0wsCaiSdriVyKQdwap3Nw0YJyafNKxpDXRlPzmnfGTVuK1Ix3rKq23Y2eisNu4Sq55rKklKnms17ujNKVpaDopt/GcVdtovMGdxpuTSsjScbItpGUp5UHrWSbaZjuIy4Xgc02Niud1a+0s7IhLdCyT5HBqjNvZs021YdrMIo9x+Y4pJ4Wz8vSsW9G2a35WRSWjIpJrNuVck1pTm+xrCSkyqAxY5zS4wa6ltZly02EkPHJqjI2wmtIWizK5Ax3Gjbg9a1jBOWgr6DJE75rOupirYHaumnFc9mZykiOG4cvyetaNuQRuqq8bIlO4TyH6VWYkmsKe2oSfQFznJFErA/WtY72Ila1yFnK8ZqvIS3Jq0kmZuzQzaTzUbNt61a1JfYmtgJCAetaC6E9ym6MHNZVqnstWLlbGjwzdnjyzVmy8FXM7jchA+lYTzCEVe4nScnf8Ar8zp9P8Ah+CoDrjjrWva+BoImB2A49RXz+IzVttIcYQizasfD8NsMBRitAWUadq8WtiJTdzQNiCgg56VlfuK1hQhNMliOOmaE9RNkC53YIxUjQ7lz61tbWwpPS5UaEqxI4qrOmeua3g9RqVys+CeKZICymuhDKcyHHJqjOMciuyErjRWE+1vetOzn3Lg8130Ykz7l9JduAO9TxuzGt76XMOpdhfAHPNW4JD1rmqt6saWpHeq8vAzg0yy0qYtu565rzIynJv/AIJ306ijGx1mk+ZAo3Z655rft7tXXG4E10Yf3YrU4qtm7k+5WHNOhiTdnFdCSnJM5noX7YA8dK04NOW4HzDOfzr1I2aMXKxW1XwuskZwufwrzzxD4aNqXaMCvOzCi0lOJ6mBr3fKzlLmxLkgjNZN3oh3bgOteYq12rnsxdjIn0llc8Go10tgeQRXX7RpWZFZXLUNkqjpVhQqcVnLUVKn3F80A8U8XFTyO1zqUQ+08/eoWcdc0nzctzTlENzx1pBesO/NaptO7LVNMkjv8nBq1FqAjOVfmvSw9SLaf9fmKdG5YOqysn3ic+lULy/mQbt5z6V1zb5uZHGqEeawy21aaVgCT71tWl0/G7pWNVN6/wCf+ZzYnDRjsX1uTjvVm3uG78CvMTd7NnlSjYtLc5HWgS7utRWbtoYtWGSRJICGAqvLYRt2FedJ23MZXEg0lCckZNW2sliT+dTF8sW2v6+4lNlc229uBU0cfl149SpzHu4CKsTqSwq7bQs+PWso03PRHoTaSsXk02dhkKePardqskGFbPFe3hcNOjaUloebXlGcWkbFpMxAJrSt7gjr0r3KVRvc8GS1NO1nyQfWte1mGAe9dtOVyGmaEE3cHrVyAkjrXbF9TJouREd+1SgZPFbpNq5JJGxzg1MDznNVFPlFcFI5p6ms23cocBx1pM4pyfQQoNOXnrSlF2AkUnqaXP4VcW0IQgtg5o6GqnrIF5BnimMwA5o3BFeWTaD7etZl7JgZJ6nvSd72B6nHeI7gvnpzzXKXL1oo2iVGxmXAIyetVJPWqvcu5A3fJqNjgU2O5ExPemkg9KmwxjNz1pjHmi3VhsfJjxEEk8Ci1QtMo5xmupT6IcV7x6h4WAW3QY7VvzS7EzmsnKyZ1LR2MS+uOpyawb+Xcev61yVHJ6o6Y9ymHOaXaGNckeZvU2s3qi3bxEdc1b8sYzioqO2iBu4gHOKVkO2sW7FrTQhDHd1q/bTMseDVSkrmnSxFct5p60yGFw2e1Dl2CDLSZHWpMjHSsFdsvzGY55zVuBgOK1ejQbjyoJ5NWI4vlyRXNNu+g27IXaVGaTcx4OatxutTJMesm0804y55JrFa6o0T1Fa64xnFRpmRvaod0xpF2KJSuDUV1CMGlFOzJk3cqfZSSe9RSWxQ5roTto0Vzalq1m2pjNSMGkbPWsq0k3puCVncsxQ/KCaV/lNClqYTd3YZK2V61UlVieK3U76smOm46JCOpqwBwOa03WhLl2GmUocZxSpcM7Y7UnJ9RpaXHc5zmkaTHeiT0ViCNrjFQtIGzRLa5dtCrMQTxTUtPN5p8wN2iQ3FiUywFRRwnOOlUpdieYvW0W3nNacA+WsqsncpvQspJsGc09bok4pfFGxNupOsoIp25WFHJpYkb5YDZq1Eg25pKLvqJvQguDhuOKrS3LIDgmqTcXccddyq1y7nrUsZZhyaiVS7tY0dkOJYDk1TuULmm5IStcokNC+RWnZTGVeazb11Nm7q5LeFTFg1iXNtk5Arnm2a0dCv5MiHgVoWJkHUGqUuxvUa5TR8zjmojIS3JrTkXJY409R6yZXqKjmbI61nGKT1KtYrgkN1qdYfMXrRe2jCXcUWpHQ1JHAV681PxLlsS5XK9+5VcVkMpYkkZya2hFRV7F09iGVBnNQvHuOatt6GtyObCrWZPKASM1tFKQk9SHgnOaY82011UYu+gMas28Yz1qtJbF2JFa6p3ZlJAliwOeTVmIbBilWqXVhR00HugIzVdwM5rKDuTLcZvA96iY5OTWyWoEbGonPNXFPcykxu8ge1RSHOc1UdyWtRkcpSQEGu+8HvHdIobFcuaJ+xui6aT0O2g0iBsHYDmtCDTIExhAK+Gq15bXK1LflqvGBSdK5Lt7ktBnvSct1oBjCpJ70oTkZzWlna5HNdkyID1qRwgXmilOMZXkrkyu9inMEByKiacAEV1urFzbSEou2pXllBPWqtx8ymqum7pFoo+WQ3pQ529q61JcqQ2zMvpiCRmqpJZOa6oL3S+hVkTD1etG2gV2Umr7kSLyPnBq5A3GSa362MWi1G1XLdia56rsmNGhbIHxurbsreMIOBk1hhqUbvzJm2XWgyh2dRSWsciPzkZrZ0eXVbEqWljZt03gc1aWEJVRXVGcn0LdoVDrx3rqNJVDHg46fnXpU2kjkqaF28gHl/h2rhPFVgrbjjrWWKip02jowk+WZwOo2JRyduayZouSCK+a9jaXvf1+B9HTlzIrSWiSD7oqpPpqDn0rotZaf1+BsnZWMi+IhYjNZ0l8Ac7uBWlODep0RjpoRDUh6046kBznj1rr9n3/r8TWEdRj6qoH3hk0LqS4+9UqhK1zdQEOoD14oN4pHX9a09m72NVAjk1PacA9KE1ItyXx+NdFCFnr/X4msoWRo2usKqgEimzX/2p8DpXoSgrHCqD5nI0LCNRgsuRV8XGxs0uS0NTllS5mzRtJvMAJOasyShB1xXi1qfvXR4eIjyzsEVwWPU1bickVy1W0+U5JE4BNKVx1rzqurSRkx8V5HEdpPWpJZllHBp4u0aatuRGLcrEagA0MpPOa8RJvU+jwlPkiPhHODW3p0iw7ScD6969DL4pS5pFV9Tdi1C22YLAEj161BPNFK3ykV9JPE06sOWLPLqxcSxaEHHNXlcdjUx00PJl8Ratrja3XpWvaXYI6n3rqpz2FJamraThgM1qW7qy5GeK9CnJ3sYy3LsBLEAetWlQg9q66ej1IbJV464pcYNaNO2hInKHvyc1IGrJKzuyrjw3elIB5zT8wBcMTg4pyk0Ti+awIcPrS7uPWk33AXdgc4FMZuxbJqnK70BIQyEEc1HJJySaprQnqVJ5QB1Gc1jandYU/1PNFnzXEnZnE61dmSUgHjvWHO23ir1NbFC4+YVSfOatNJ6DRE5qNiMVJREajNLWwNjG601vrT3Qr3PlQjNTWSBp1rRabnQo6no/h5dluvGOKv3kuFx3qZ+Rul7xjXcnByRWNeOGzg1xyep0R1G21uZO1X4rRVX5gM1yzlzSNbvoTJGoNT4G2omk9BeQwIA3SlmACUnBWsU7lJXHmVdRgU4pTikrmvQWKIO3zVehgUJ0rlnuWloRumDxUbZo5uVal7DQrE5qzAh6kVrKaQt0WVGeTUqyEVgpJSsS/hGm7AboKGlU8966JQ7GLjYqyznPBoWZj70WjDYtSFaY5xV6yXcAcmuarq7miehoxpgVFMhBrNIylLUhLAVDMoatlHWwIhUEMM9K1bRVZRwDWU0r3ZpJ6aFhwAOBUJiLnmlO0VcyWoht6Ps+B0pU53VmTLzInhwCOKrO7Rn3rqi7Epa2Ywuz9etPj3KOaG7uxTtsP3vTZQ+OnNWnpZkaXuVZQ4PrTRu5o5tCnqJ5bMeavWUY28ipvfYOhPNbRtyaovbopyMVUY21Mle45IyeB0q3FEwHWpk1f1Kv0JgvHNNyN3Whp30BXY7zMd6etwVHWrt3DcPtDE5zUyXbKKnns9AsthGlZzmo5YzjoalyTYFYkI3IqxHIm3NKMItj5Xa5FPPtqIXAcVMUm2mVy6XKs672qxbSLGuM81TgmN6qxHczMz98U+OPzVziuasklodFN6EZtTu6VetYBt5FVShfWxNaeg6YBapyA9eta1eWLsZwGAt0FPWNm5JNZSdtTS4jW/OasW/y8Gue7crlPVFggVKoUDJFawjqYO5naigaqIQelaPsjSG1irPFl+aimQKnFKEdLGhmXLHJyay5hluBXVS+G5SGZ2Dk1DLJu+ldtFa3IkrDY8girsS4GSKuq4kvYeGAOcUu1HG7Fckkt0JX3RWuG5xVWViK2ppWBkBcjmk3E9a30TJepG5INQyMeo61cWmYuw0PkZPWkY7h0p2XQTQwriuq8E3vlThT61hjFzUJIqG56nYSiSMN61ox9K/PKytJmshx55NN561gQwyfSnKe5psgQqScigU+Z2sToSimyAkVHUZWlQ96ryKMmt4sm/YrSIfSqsisDmuumx3GNgjPeq8uQDW0dykZlxtdjuqLaFFd0ditSCUU1JSnWuiD1Ka0L9tcA+lXopckV3NbPqc7WpcickVctmya469rWEkatpkkVswSkLxwawoT0s2KaLkN1jGever1uRLhiMV1811ymTWhftztOKuKN65rZQ0sjJsmt7dw4JBxW3YymMjmtIp7MzkaMl5iP5jmuc1crKTzW7tbUILldznbnS1mBwBk1iXvhhnJKqQfpXlYinz7HpUMRy6Mybrw9cRfdBNY2o2d3bITsJ96yjTlFeR6dPEQk7M5LVjNuO4VhTyOQR0remlujqeJitCuQx4yaY8jDgmuvRsyhjU3/X+RVluCP4jSLduvcn8a64RutTvp4m7t/X5En21gvLUz+0zj71OEN3/AF+Z6FKakv6/yGm+LDj9aFvXI+9ir5OV3NrluznmlOM4B71v6ZbnAw2TVSetuhnUskzZRzGoXNKGJYdTU16iSOVRVjUsGbHtVyRGYZrzpWabPn8crSHQQsD0NaMMbeleTWkr3PLepOsbk9DUjwMUxjFct10MZNGe9pJ5nBNW4YmVea4MXVvojWhJOaJ4bdycmppI9o6VdGgvZXkfQpprQbB8r81b8wbetbwhG1omU3qCzEHOasRXm3qTWtKShoefiHdF611AqRg9fWtCK/z7ivRjNWWp5koFmG8+YEZ61p2V6MjJNdFPfQyaNqyvMgc45rYspvUnFd9KTuYM17WTOOc881dSTPBJOK9KnJbGbTuWEYEcdeue9GOP8a1V9iROoJNLsOM5pTSGmADZ9qkBwM1PJoO/QM4OTxTkk+bPGBUT3uw6Cl8t7UjvzWdtbjSBs45NRlyOc/jWjS3EMaQZHXFMklwD3qotsRQuZTnIPWuc1u8EefpVvfUFvc467nLvnpms+ZuppmiRRnc55qrKc9qpWKsQMxpjNxzSeor2Is4yaYefrSV0rsGMJ4pjdaq1gPljGasWCEXCdOtF22dK3PRNHz5C/Sp7tuKJ7aHRFu5iXr4JGaxZ5syEbu9cc1a50R01ZoafMir8xq3JPleAK5KkeV3NEtbjYTI7ZNX7eEsOaylNtmnKhLqIxjiqEszgEE1pTs1ZmV76FIud+QKvW0uUx3p1PdjY3jEspuU5FXIp3KYPFcl+ki0lYaW3PzUoty69cUpJ6PoO9rMXyNvGRViKPj0NYqXM72BvS4yZip606Es9Xo3rsJq8SC8iZfmqmLlgcGtYVdGZpXF87dyKfFLg9auUtDRQ0Jol8x81p2pKLXNNLZCl2LP2n3pDIG71cUYtakEynPBpUTIANKUlDUvoO8kGrNuSuATUzhdaEXvoW1XcM1E52vxTnBNEwepKi76V4jyc1yx0nZGkyndfu0rLmnJfBrugroiKvqTW4MnJq2qqq8ikopEy8hhZQcgU7crjpW0Ipp3JsQyxr1GKRIAwqZxWyGr2HGAL1FLkoOKzSsJO5HJOxBzmqbysG9q2V2UkSJcFTVyC4Zxmlp1KcVa5OHJqOT1zWUm27olaaArE05mx1NJ3YWBXBNTKN1Fr7AXbaANy1WJLddmOK0jRTMpN3Me8t9rHFU/NKcZrm1jOx0waasIzGTv1qdLcbc5qlJ8xMtrDJYRjrVUKQ/WrU2hwZoxxpIgyKljhVD0rGUbyHqh8iDGcUxZSgwK6Yrk1Ivd6leV2dqChK8msqlm7ml0LFFzmrS2+5elczbd7jb6iNakdaBFsOcVnF2G5XHD72adLJheK3vd2IaMy6dnaoSQi5NaKSvyltaFaZt/NQNDvHJqra80SpOyMvUIiM1mOp9K0p7FwIZIyV4qEWzHrXfS8xSY+GLZnPWrUMbfxVNZroLvclEWOtRSjb0NcqfvCiytKOM1XYFq6I6bktakbQ45puAK1vfYztqMZCT1qGWIr3qlLUh7kW3nilIx1rS5LGsM81c0a8azu1YHjPNKa5oOLHFnq/hnVVuoQM4rpYiCOtfAY6m4VGjZomHTPagj0rzzJjec0qrmmJj9mRSBCOtK4vUlSMsKGjIHSpuJ7kEyDFVnTHWtoMCJ1BqpcRgZrqpycWKJTYYNVLoPjINdlNX1NFuZs/BOTVZ5sHGa7oK5pF3GNKMckVA0ozwa3jFlSRJDIc9a17VgAMnmu1Sdkc0l0NKE8Cr1opZhzXDiLJMhK2p0FhbkgHFaMcQBrkUbMlllYQec1bttwGOuK6qL10M35l2FmBGa0baYd69CEl1M2i9HcJ27VINQCHk4q5TiiFFhJqZfgMSB0qpO7PyTzSnNNaFqNiBV3Nir1vaI6/Niop2uTNtBNp0EoIZRVO58JW9yp24yR3HFddKkm7E+2kkef+NPAaRxyMsYVsdQK8h1ezks7hkYYKmuKtQ9lUv0OlVnKN2Zcku3kmqVxc8E5rWnC+pph9ym0rM3NAcryTXatEezTFMhYUzDOck9O1VFdT1KMrDtrMABkU8RFDmq5t0zvpS6GpYSche/vXSWEqIgA6+tRFDqQdiz9rVTjOat20okI9a56nvKzOapFxjc3LGP5RzWisQAyTnNedKXKj5zGO7LECAmr8MPSvMrRTTZ5kkXoLPd1FWW0/cvSsYU20zGcijPZeWelQqgBwa82rSbnyy/r8CqV73LUUeFzTZU3V6mnJY92hK6IHQIM1CHO7Ga5VKUXoaT2uSiTjrSq/wA3WtZS95NHm1NS7bMSfatGAnGK6qfvs4pItJuGOcVainKkHJrsjdKxjujSs9R2MNze9dHpuoBwDuHriu6jI55rU3LO6yAN1acEu7HPWvSo1Ohk0X4CPb0qY8jk5rsj3ZmxhbBHPFPXmhvUY8x4HJFMDYoYhHYe+aiExzXNUbasaIcJsdxSq+5ueaiEveswfkPLbhio2z9a0loTfQgdtvB7VBJNgd6pNxVwZQvp9seS1cXrd75j43cAnmtW7DjZmDK+fX8aqyt65poooXDYPNVpH4oS1KIWJpjZbrTatqCI2ph+tJu6GxhPODmmMfSiS0FY+WyQBxVjTSWuVJx1rRNX1OmG+p6FpP8AqQe+KfeOFB9aio+xutzEvXUg+tYzRmWXArmmlctv3i1HayRJvqzZszyYbk1yVZ7nTTfMbtpY5TOOtWhbiJc4xXC5v5lMpX0vasmb524rqp66ohLXQWKDuaswWhLg1hOfvG9rK5fjtyoqQRgD3rJrmkik7gI9pzT1nGea0UGroa8hyHdIOa0IlAXmsXHoKb0Kt3DjLAVHbHB78UNq1hxd4ktzhl5rJlhXfkCiEXdoUU9RBEB+NPSLGO9azfQ1j2Ldsp3VoKQqVnuzOrqMaQE4FOjJxzWilrYyWg48mnqSO1TKKb1Fzdxy7l61LGTnJp3S1IdnqizE+eM0kqgcmnNqSuhbMltpBnGasOFIyCKxdPaSHJsp3CIQcsKz3tUkcnrWig0yU3uh6xeUKU5NbJLqK43AyfWkY7ec0bBZ3I1fJ5NTrhRnNODVgluMNwD60bww60rJ6MLW1IbhRt4FZ0kmw89qfw6IqLuNSYselaVqxCZrOb1sjSSsrFnJK5FRDczYNRza2RK8ydE4pXh70LqQ3roCxgetTq6oPaly8oX0JkvVA609r7K8VvCXL1/r7yeV7soXMjOeKpSRsfesKtpPmRqmhIwVPNW4gzCpUlFXB6khsncd6glsJEPApSn1BSVx8McinnNWkJHNYOpJyuaSs0JI3FQFt1dHO7JkJdRwUKKQkY6VHNfUaXUktVJbJFX1TGKnS9kKUhTgnmopUqYwTFcrNKENV7i5KirTSepaVysZVfnNMdd3Ss3KzZaTGrAWWql24hBq6TvIXWxlTuJWNVJLUsSRiuu8UVqCWZIyRUT2+04rWNVW0G9xohCnPep4lGKzlK9rlW0FZCx4qCaM45oi1czKkvAqu3B6103RL1EOcGoSfWtI+ZIEgDIqCU5PB5px7ktkQGWpr5B5rYjqNU560hJDZFXZbE9TqPCviM2kqxs/HQZr1LRtSW8iXnnFfJ53huWXOjVM1h0zRkV8wS9RVTIp8acUmyB2QvWpogj4rWEYtNSBssqiDnimTKCO1a+yhyWRPmUZlX1qpIFz1rOCXMGtiB2RepFUbudecGuxxTloONzOe4UNywqOSdZRgc11wjLoatdSlcw7gSBWRdIyGu+gn1GmZ07zL0FQiSbqc16EYxsauStcuWUzkjdW/aEhQ1TO/wBk56ljStySByTW9o9r5jBjXFiG3oupD0R0tvGI0AqUIWNYqDdkZF22gJ61et7YBscV1UoO9zOTL/2YbQfambMHGa7ZU+XVEKVyaNSB1qUoSO+aTfNoxN6jY4GDdeKnMfFZ2bKuNCbCTig3bR9Kyi5Q3BpMgk1VlPNW7XW1C/Mwx710UsTZ+8yJU77GR4s1O3ltzyCcdq8J8Yxo9y8i9c1tiavNZo6MPB2scPfvtY4zWf8AMx710U42imd9KlbckWI//Xp3knFUnc9KmtkOW2aRgo71uab4VkuEDsOtUrvSJvKapolufDD25JCVnS6YUJ65qbWev9fgd+Gqpi21qUfdmtOGR1Gc1nVrW2PSVpbluBWkOTWxp8fzc1z892cWM0VkbVtcBOBV1ZmbFZVYKx8riN3cvWW4uK3bWHdjjk15laKtY82W5rWduByQDWgtuNnSpjZxsc8zM1ODaCR1rCdikvPrXnYiHs5ps7sJT50XYXBXrTLhwozmtpL3dDsp3g9TMubllPSqwue9cqbvqbzmmtCSOfJ61ahYHk10J32OCd2XrdgD1rY0/D9cV24Ze8kclR2Rp+SpXPFM2Y6c16FWHK9DGOo4BlORWhYX5iYDOKUJpMJJPU6LTdT3ck8DHU10FpeLwR0PvXo0pLY5prU1rWfcBznNXA2cV61OziYNAeetKr4OPzqmhLsSCQN3/OopJABmomNIiZy3emnOK5ZItMRRznNSKTnrzWaT59BNj1YmhnwK6Lakladvc/41RlmA5znPpVRS1Qne2hh61qIjjIBPJrjbycyOx569KtX6lQVijKxzUEr8U7I0SM+dssarOecGhbjSIiajLEGnzaMaQx2xTCcCi6SBoYT3zTSe9L4kDPlk5PGauaSmblT15/OqUrnRF6noOmA+QOe1Q6m5CE1lV20Nou0jmbm7feQadYANJlq425Wuzojq9DYdQYvamWCr9p61zOF27m0dzqbNUCA0tztJ6/hXLNNyTJcruxjahAeeazTFj8K3pe7HU0g0WIQCBzzWhaxADPBJrOcVctu2pcSHdUcsBU96y1jMSlqNKHbzVZ8g56VrKrzOxSfUdFKcg1oW8zOvJpSklqOa0HTElcE1WLiP3NZtcyukTDsI1wDwaqzkE5reGtmUtyNQW5zTlVg3JqKqd9OhrGSRpWaDAJqxKuBxUN31RhNu5WJA70iXWDgmqTjZyIUWyzFMrEVMzA85xipi7pNhOLQgywPNSodoqnq2TYnjHGe9Dh261KuS33EJ2DOarXF9IMhWwa3p3UbMlO+5nzXk8rYyauWLPt+etJyvHlRc7JaFtyGFV3B9axd9DP1GKCKbIrMKXM7tFX6kIVg2TmnsTjr0oTdimxIxvbk1MU2jOaNnclsq3DNjrzVGUdya0jLuVFofapuataO2ITiuSejuinoGx1NKsbscmqVRX2FdbkyZUYNKzHrWid3YzYzzMHmobu6CgAGmtXqXFakEUxY+tXI5CVqZX7lTQ/qMnmoWBJ6UrIyuLFblm5rTs7MHlhU8nMypTstDUjtkVenWq9zbL2FVUglHUxjJ31Khhy3NI0eBXKoaPQ3TK0opm3A4FaJ+6WhDkigKTWFSWtirdSaFsHk1bDgjrWlLYzkKWUDJNV57gAda2jtclXZnyvls5qrcvn+I1LV2bx1GQqSKeVINYuLa5ma31HebtXFUL2FpWzWlLQhrqUJbQo2aRVA4NXfmaiarWISSLGOtU2kDv1rdxs7LYSi9wMXcU+OL1rObsxrYe0eBzUckIK0oyvsZTMu7jIc1WdNvJrthJaIjoMZ6ru+T1rdakDXf5TzUOTmqijO47bjk0x+e9Uib6jAPemsDnitE9QYLI0bBgcYrvfA3iRiyxM/I9T1rhzOiqtBjiz02znFzCG7VOFzxX59ONpWG5E6Q+9JLmOrqQUEmjFyIGkz1oWbZ3rIy9qK+oFO9QtqhK8mumm2lZf1+JXOVJNSPeqs1+zZwRVwp9Re2RTlvWJ4NUp7hmySa7IQ6sUqxiX166SYBqxp1z5mNxr2aFJOJ1KTcLmoFDjnpVa4s0btWlSmraEJ2Kk2nptqjLZqtZ2lHf+vwNItsS1th5nSug0/TpJQABkVTnLl0Im+50en6H8oyMmtm0s/IGK45pqXM2ZOZfiBPWr1vF3xyKuK5ncmTLUXyHmrCXSJzxXZStFambD+1W5GeKclwznNaVK19ENRSL1oxc/StCKDdzTirpMyqNIV4ivTvQqfLz1ptWEpaENx8oOKoO4Y9azauaK5UuztUtmsa71VoARuIrJQ960jamuY5vXtYmlQhWNcDrLNJu3tmtJJ7NHpUIKK0OWvINzE1FDZ89K6lJ2OyNO5bSxGMtTTZnPtRFtux1U4JPUntreOOVWb1rs9E1C1ijBZhkdOa7MPdbk4qnKa0F1bUYZwQu0D681zV3sdjtOaVZRT0/Q3wVNwWpXSPnJ6VZgQt0FedVktWe5B3RpWdruYE1u2dluHf8Kxp1ObZHn46dkXY7Fh2q5DalME1U6nus+UrSvI0bKPDg1v2uBivOqrRanE9zWtR71oxD5amn2OeaKl7aed7Vk3GiEgkda1WEVe91qdeErcjsZN2JLI8g/jWc+tIZNjMM15NVyjL2b6Hq1Ypw5kWoohdrkYOar3OlyRnIH4VHspNpo8xVtbECo0Z+bIqxFJjvW8XaNinK60L0EwBGTWja3ZUgg4reFVxSt/X4mElfRmrb324cn8zVlbgE16TqRnFdzLlaYplBHWhJMNnNYc3LuXymhaXrRn71bumatgqGbJFdVGt7ysYVIHT6dfiRAd1bNvMGFfQ0J6WOGasyVpO9MacDnNdM/huSlqRvdEDrj8cUgn3ZzXEsR71mauI7zgOPWnqwfqa0cU7MjUkCZUU/wAsgdapU0gGBj6frUbHB5PWpaYmQTE9yayL24CEknnNVB6hfojkdbu9+QGPH/1qw3lHc1tHzLSdivLIrcjNVpDmoejKehRmOCecmqrEk5zTt1Y0xjVGx5qdwTGHnk0xiaa3Ghh4pjZ6ml1EfLanqTV7SmzcLVP4tDpje56DphH2cfSoNTG8dTSqG0dGcvexYkJ96saeBwSea4pO11Y6KbvqawG5OSKjt0CTg571yO/Q1judDbyZQYPFQXUzJJkHis3pK4P4rEJnEvXvUElsSC1QpLcp+6Ronl8mrEFycituT7TLiuY1YJlCbiKSWZXOawnG75ibO9yGR+5qnJ8xqFo7m0BVYLzVu3mAxzVct2W43H3EoYcVAnzjBpT92NjJIkFtkZHWop4MdqzXNuK+pFGmD0qYJjGa1vpqO/UtQHb1PFPlmXbjdS5ewpNsqO2ScGowCWwetTNWjYuO5bihIwT0qR3x+FRCdtwnqLbz7jirYO4VtUd9jGSsyeIlRmnmTcMVMHfWRm9SORdwqnPDjkirbdtBbMjRVzjFTBWHQGrVk7h6kqRsRzQ8eB0pKV3YUiuXxxRvJ71OzBrqMY45qvNPztBzW3L1HHUIg/3iama4BGCaa1RT1IHcFSTVSUgnpUuO4ktS1YQbyG6Vqp8i4rCXYqUtQz83NOLhc0JLVJGbITcAtipVkXHJFC3G43RSup8EkHNUGLzNjPFU2kjSmi1bxle9W1dVHJFZ67jnrogW7UttBqXK7c9zWkLNXIlBx0JIZADzxWjBdogGSKqDTkZTi2WReq3AIpklxkdadWwlEhzk7u1JIcjJrm2NEVHGW60bRis1Tdy2yMYB5okcBcgisZ09bmpCk3z81ZWXAzmt4NWIlca85A61QmvGDYrock4igrvUglnyOtRRkynk1mjXYkJaKmtc7qmeisOKuC8nJNLIVK9Oamm76FSizPuDtY5qB1ZgSBW3LZXKTKMrFmIPalgt8nOSTV87LuW1tzjmnxW+WrOo1bmRNyG4jZWqJ8hMk0qV2tNzKozOuVGc1SuJM12xjsSldFaRjioDkmuiJnJ9BrHtmhVOenFX0M5Ow7BIphhbrihOxNx6WzN0HJqT+zJTyAaTnbciU7Dv7ElYdDVjT7KfT7hZFzkGonXjKLTM41Vc9T8L6m0lugY9uRXSo+ec18HjafJVdjoaJTdLGMk1BJeCQ8YrNrmhc56uiIWkHrTDJ71mkcF9Rkg3LmqcuV71tT7F3ZVfJbrTH69a6kS7kRTrUEyDaa1i9RpnO6pH85p2mSbXAr2qEvdR6dN3idDFKDGOaZO5HfNEpPm1KUSjcTnqKoyTFjVuSlGzNEi9olkbiXJGRXbWdgsMSttFQ9Ec9Z6mhbuQcAVeTkc1zS96zZm9C5apuYCtmK12RgkV10qfu8xlORXnB5xVRQ5b5ulY80r2RSZajhU4Jq5BGG4rZLWwmzTtItpGK1bZMiumGyMJln7J5nUYqGazZPWqklZiiyq9qXJrMurJ4skKcCuWalbmNYvUzblGwQawdQtS4JqXLm946aZzWqWhVTk1xmsRkMwxVqXN/X/APWwzRgtFl8mh2ROwrVK+56myuQzXu0EcVVbUGzgmuulBR1ZmqsWxwvS2OamS+lTGJOPanKfLLy/rzO6m4tE39oMxwWqVZlYZ3ZrCrUk9b/195vCPYnQCSrdupzXn1JXvc6oM3NMg3uucV1thZrgHFXg4KSfMePmNSxNLGiHp0qIzJ0PFZ10ruKPnZpt3JbaTJ4NbNnKeATXDUqNrlOeSsbVpKB1NaEVwuOvWrpNbnJUXQereY3rmrkcCleRzXsYZe7czjJpmPrmkJNGxK54rzPXtEu4LovErEZ4ArnzDCxlaUVqerSre5Zmx4WdvlWUHPQg12K6bHcQg46147uvi3PLqu0zD1fRjGGKr0NYLExPhj0rlrSs79zspSuieGbBzmrcVwc9arnfX+vxLt3LsF0QRzWhFdEqMGt4V29L/ANfeS4q9ydZx60ou1U8nNKVdN2v/AF95qqbaB9RKD5TiptN1aYzD5sgdq6qMm9UzOpS0O00TVW+UsRzg/Sutsr8OBzyRX0WAqOUTy60LMvLL3OcVFJIT1PSvRrX5TGOowDceTQcjpXmuD3RrfQA5JyTUsbkEelddGV9DORchkAxk1PvyK6L2II265NRSHjrSemoIqXTlEJzXJ61f+WWB71KfUa1OWupmkYljyaz5Ac5Na3LRCfc1DKxApXu9SijO3zcVXYn2pN9GURHrzTS3PSluJajGOTTCcnij0AYW5zTC3WlsOx8t+Wc5zV3Slb7QD71rfqzeMveR3enzhIhznimXsyup55qZy6nRFPc5y+++TUVtPsfk8VyPe6Nobmn9rOzNOgnJPNYyjozaLsa9pffIBnNJPcb2PNc8tWDVncSLJBp7S7QMmseR85UnzFeWcHoaS1kJkzW6VtGON0aYuQiU1btSeTWfLfRAkwnu029RVL7SCeSKzlBm1LTcf5oI9ackpyKm7tdFuWtidCZCMk1citwwGM05yulIzbsXIrYge9OksfMrCc2uhle+oR6MGfNPm0kqOBWXtnzf1/kDZQmtJYW4GRVWTeCc11wqXRSauLD8/XrVqO3C4YjkUqybVkPmsy2iApUM0ANc/I4oOZhHCFNXoUBFaRfQym3uSSAqKjDY71o7XsZ9BfMUnrRKgYZzVa2B3RFBAperYiUjGKTvaxMm9xdgUVXuGwDVOD5dAvcz2GWzzTsgD3qo3ZbIZGPfvTFjBbPerTvoGxITjiiOIFstTb00E9EPniQLnI+lU9isxzUSV9EOF2X7QJElTGTJ4rJrohvVsUvtXc1VZrvsKa0lcFG5CpdjnNTqx24NZqp7xbKtwpJJqFX2mtJMuCuTLKW6UOXPHNRzu1kUojoYzvBOa04EDCnd8plUuSG3KjNNCsD1pRlbYzTLEO4d6kMuOtJyfNqFx3m8dajebPFD1Go6jV5570ODjiqim9RSepRuGfd3qPLkc1lN+9Y2jsORNvPWrCuMVMFrqKTK9xcAErmqkzZXNazty2HGLW5VJYn2qSM4OTUxujRvqh8rl+BSG2YR7jUTTnsOL5dyOKQhsE4qcLkZzSjBplzZXmtwzZNJ5SBdtdHI3u9DK7KV1p4zuWmRQmNsVEeqK576FhRjipUXyxuJqWm2rg3ZFeeVX+tULpsA4PWt4xUHzIykzOmP51SmB7iumLu9ReZVbd3qJ+2K6VboZyeo+1jEkuDWwumqUB4NZ1JWOepNpj4dIDnmrQ0FcdKzlVV9P6/AydR9CxbaEiEHaK0k0aNlB21lUTkua/8AX3ETmxf7LSPqOKrTWcYboK4JVenX+vIzu2zR0iUWh4PFdAmsALyea8/E0XOSkjspyuhkl9NcnCZ5oVrmI5fOK5/Zcsb2MKzvoWI7rd1pzSnOa5ZQ1ON2THLNnrTJUzzUpWZcWQiINmq88OK2i9RSvcjxnrUM64U+lbx3CNzmtUbEhqnBLsfOcV7VDSKZ6lK6ia9tfgr15qWS7G3rk06rcn/X+Z0JWKss4frVVyC2amN0CRs+HtQjgkCsQOa72wmS4jBBHNaz96NjlqqzuX44VHIqdE5rj62MWWLSQrOK6D7QGh5weK9jD8nsW2YzXvKxSLBmNHlA81xKmnqXZoUKV6VatTkiqhdyB7G3aRlwDW3YWikD5a7YRVzmkzXitV24wKr3NmCCP1rSoklYiL1Ka2YDcj9aWfTVkTgCsoQTVmaOTuYV9oLsCygVhXuiSZPy/pXDiItP3Tppy7nOat4ekYMdpyfyrhNY8OT+a2Rnn0rnc+W1up6mGqowNQ8OywRGVVJ9a5PUHaGQhhjmu2hJ3sXicb7tiqJN/U0jqD0rrcmefTryvcVcg80oY+9GrPYpYqyHIzZ71btmb1PNRUeh6NDF8zt/X5GpbROQM/zrSt4iMHHNebVldnousjo9AtvPcFvXpXYQW4WHg12YaHu3PCx9RuVild9xuqgyPuz1rza1W0nc4Ei1bPs5PWtC3v8AaevSuCU3e39fmZyot6lyPVinOf1q7aamZGGWNdNHff8Ar7zCpRdrnSaUTLg+lbSxjZX0FNe4rHma8xTni3H5hWXfaHDdq25etTNqWjOlTaKEXhsW8m5RjFaNurQDaTkV5eIo8qujKrq7hdwLcRMTnOK4jW7IQzEjjntXDWipU722NcK7uzKEYPrVuINXHyuTO12LluCetXYyVHGa6VT5UxpX0HmYjuaYZWznNcdmnzI7YQ90kiHmn5s1taVZxBgx44616eEi+pliI2ibkDCEDaelbOlamwlAJ6V7OBnaWp4lePVnUxXW6FWz25pj3YJ4IIr26jvGxyRQ9Jw3JqRZA3euVRbReqQb0U8mpEcE8/lRC8Xcl6k6SVOkxPP6V16Ncxm+weaCST3P4VE78HFJyugsUL9x5R5rh9dlzLjOaW6TKjuc/M/zEjp/Oq7uWzmreupoiJ2qvMwC0r9hrcoyueagOSc5oS1uMjb601jQ2K5Gc01jjvSGMc7RkmmEZ5o6XHbQ+ZUUAdeKs2JAnXFO75tTZR1OohlKwjFVprptx/xrCtUsdtLUpXD7wSapJkPkniuaEtWxuNmacDBk60rSlOAad7suxLFdcDBNWrYu7gkmsKkkrs1ha2ppxBV56VFdkBck1guaVmEkZjyEHrxVmBsjINaN2epaWlxzTMW254pSxC9aFpK42QSzM3tUaZz1zVOSsVE0IOVqQIQeKxSstQ66ly1HPTmtezgz1PWuZytdPcmZoCABelN2/NWbsZpksLYPNTE7hyM1k1GzK3RWuY1YHism5tRyeaqno7sl6DLazIOf51cEIAwTXRfqK41oyo68VHyDzzWTm3oUh6AsMmp4/lGc1pGKTJmJJKfrUTOzcVck73Iil1GlJFOeamR+MGtISbWo27q6BRhs5q3ByKEr7kT2HysAprPnkBJzW1lYiKZVdxzk80i4asE9dDRrQVoSfmNMwAfSravoG4xsZo83HQ0l7urKsG1puuTSGEIeaI3k9BNtaDXmZRUkU7NUvTUpJBPOcYzUAO45NQ7yRcUWogMUkp2jg1EE76kt6lOabacZNVjNvatalrGsCzFuIHpViNTnniuXm941aVidMVctiQRmt0n1OWe+pbZsjtUDYU5zQlaViPQUT44pxYNjmpktR2JC4C1A55zWvS4k7DlcgZoebiqptdAaKc0oU5JqM3SGs6lNNpmkbtALgEdRSmTKk5NZ2adh2KyKZZOalljCDBPNZ1IuTsasr+WM9aVI13cmtorSyId0MuCIm3Kc4qN9QDR4JxVSVtFuaU1zK5WE4DZzU63Z2jFZ8zNnC4omEg5amgZfg1TbaOe3Kx8ygJ1qtjHWqS5UIUYHvRPJ8nBqLOQ5bGdI5B5NVpmDg57Vsk5bENFCRhnrUL81ukOS0KsvynmoC46966I7GDWotqxEwPvXRQSboxWddaWOeqWreUxsCa04HEoya5Yb3M1tdFlcA1bhbHWiUtCZK4+RBIKz7mDBJriqxs+ZIzWjI4wR3q7YQyTSgc81MZN3RqnodroOiLgFh1rbufDsc0PyqK76GGjKnaaOeT1OR1bTJNOm4B2mqBuCOteFiqHs5uJnJJjoLoM2Ca0EAdK4KqsCIZE2tmq0zHPNaQinZse4+CNX681X1JBGh5rsdHTmTCK1RxOrTETGqAmBHWvYox9xHtcnuJonhuyh61O12WXqauae44ojEzEd6lhO8c0RhzGiRIN8cgKmul0TXpIMK7cYxUataCq0lax12m61HOF+cZNbCShgCCK5Ltt33OGUGtyWNhuHrWlC7sgFddGXucvcza6scVYU5GbNaJ6WSErMmXk1atUwwNOCRlJG7YDgZPNb9m+AMV1Q8zCSsXkm9KcZAeSRVykpOxNiKcrjK9vSoDKAaLJMpJtENxdxIuX4xWa93bzvjA5rOajJ8qKV0Nl0uG6QgqOa5rU/DMQlwFz+FclSnGOptTqM53XfDEaW7/KBkdMV4T4304Wd86jua0UbNPuTKSkjmUOTVlQcZNdMtjoox6igd6UJmsmzobVySKE5q9a2/wAwzWFSZ34X4rm7Y2wOM1sW9kOuMVxS1R6LmzW06L7OcqeK2k1BtmO1dNOsowZyVVzu7IXO8daiZcV5Fdu+gox7ipGzd+KmSFvWuHmlz3ZbikiwkLGtPT4DkE120NXqcOJmuWx1mknaBk8VuRyhgDnrX09K/LY8Kb1uK9qZOaja22Ck4MfPoNMXy8546VnXoSLknH9aVSipwswvdGTd6zFChG/tXK6rqgupCVwcmvCr+7eCOnDQ1uVLcF25rUgtvl5rnpUnJNnZN2LlralzgCrwsmVQcV1ulaOuoQd3Yhlh2HJqJlBOc15c4xvoepBe6SxMFPWtCzuipGDiuyhNr5kVKd1Y2ra4BQEkYqe3vBDIDu6H1r1KMkmmzxa8Ohsx+KEghILY2jsarR+MImk5kGO3Nd9XHwha73OSNB2uX4PE8LDIcfnTp/F0FvzvHrnPSrhi4WumNUm2VF8cxyTBQ+fock1uaXrCXmCHGcevNY08SqkrRJnC25tRSqV61L5nHWvWg+ZWZzNajfOI702SXI61PWwkZGqXwQNlunvXFapceZISTVxVikZMslQM3HHNHqWkQu5qncyjkdKpLsUimx560xjzSu7jGk45qItz1pS1QrXGsT60xjn61Oo0uoxjxTN2aq+guh8ylj071JYk+eDnvWlrHRE6mH5oB9KqzLhsGuGu7HbTdiCRMjjvVcwkN7VhFO2pb7jldl71YwzjPNaWTBbE1qjFuRWzaQhVyRXJVV9EaLZEzybRVG5nMgxmlFFX1II0DHHJqyPkGKmSs1qDZFnDZqRpvl962vzNFLUi8tn5xUscJHY1nPT3UUnqW4EY9qtCFgOlZtpatjla5atIzwMCtW2+XtXPJ3loZtl2N6k8vnpUzVzPqLtAOTQZAO9YKLNERk76ikhHU1aWpMhg2dBUNwxUZFdEaWhnG6epV+3EnBzUkbb+cYqKkUmb2sS7wOab5jP0opu2rFbuSJCz+9SC3K84pqrzSaRjLsgYZHNQuQMmt4pW0C4xZjuqzDNjqaqO+opq+w95Aw5NVJAGNEl71yFcpTxsDkUqMFHNSmrm26JvNGznFULq4ZGOMmtJNboUFqVhPIW5zVyyQytzWNWrfRG9ly3NqK3jVM8VTvEUdKiDaaZz2Zlzbt3Tin274NbNpuxa2HTOOppkLgt6UmuqKTdi7Gy4qOd1rJaitqUZgHY1HFbfNkmsqjaSRrHQ0IdqLzStMMcUo01J3HqPiDS4xVgs8Qzk1uqijsZuzdmJ9sbFM82SRgOeaHK7bFyq5fitWKhjSswQ8is73diG7jWlzTCfeqTbBksQBPNSvCpT3qoJp2M5trYzri169TWdOhjbg03Jt3ZrTkNjZi1XEyyVMZPZlyYRjy8nFV7qZqTV2UmVftW44zTw7HnmtLtF26iMwYHcaoXUgQnBrOPN1GtyKFmkbNWQ+wYNOKvLU29BUkyeKtwqTzV6JGU9VcSd9vU1VMm81Wj3Ml3B3wvfiqk10yg561EU0ynqihJctuOTUTT5FbxXYRWkbLU08DJNaRWhLZUuM5yKgKkjJreOxzSepbsbMyENitu3iZSBjNc2In0OSrK7sWpYyEzS2dyUfBPSualLUypvobMEglGc81OCRTqNF69R6yetJIA9YyfMrEtEcdtvfAFdR4f0fJV2X0rCjBupYfNZHZafbBMACt61hBj6c17uHXQwqbGB4m05JIzge1efanaeS5rys6pKM4zRKbsUIW2ycmti3kAQHOa8CutAY4sXNRSxg1FtFYT0KzzfZzwazNQ1AurAmu2k5fCzSm7nK6kQ7E4/GqIUdq9qjpGx79PWBLFEXYAHNacGnM4GQamrOxMIO9y9DpSuvK1BLprRtjoKKTadu/8AXY0krO4C2K9RUsKlSOtVvLQJyualjeyWzggnGa6nTNcUqAzZzXNNNapnNVg5K6NyzulmIINblk6nk1vh4ts4qisaKwBx0oNqAOldjgkYJu5GITnIFTQko2TWT0Hc07O7wRntW5YXBY8t159q3hJP1MpKxqRtlaZIzLVWtqSiISu3U0oiaSndjb6lfUNOeZMZzWWmkyQSjJyPSs5K7uh83Q0o4zGgB64qvNCzscisJO8gTRh+JLUC0ZiOMV86fEm3H29mU8HINaTjew6eqZxCQHd0qbZtHSrc9LHfSjpqCxFjnmpVh9qxlIrlbehPGgwAB0q9bAYz3rnqao9TCLozYstwxWxA2QM1zN2g+52SNG27Vcj3Gs5TOeVkXbe0ZwM9TWjDohflhUrDzqvQylWUSynhwsBgjH5VP/wjjIOMfnUxymd27nNUxasR/wBlmNuRViCPyjS9m4aHn1qnOalpcFcCtrTmMjKevNe7hJ+0ile5wTdjeiRAgzjNV5grHjFd04paEJleZgi8nFch4lvZVUhMkkVnXfLBs2pas4spe302DuAPUVs2PhSSRNzgkkV4scN7RNs7VVUdhZ9FazbJFOhDdBXNyulJxNG1LU1dNt2MoyM10sGnrJEPlAOK9GhC8NTLmtIydW03ywSB0rAnBRyPSvnMzpyoz5onu4WfPEVG45NTxTbT1rWhVtFIua1L9rdMTjNXd5ZeTXfSm3Y8nExK9xvbkE1UKOrZrmxSlJnPCQqySA/eNRTGU9SxH1rnjOTKbsQKzRPuBIrovD+uSRTKpYgZ9a9DC1Wmmc9SN1c77TtVM8asW/WtFLtnxhuPrX0lOWyONkgkPeobq7EURPf3NdXNcxZyur37uxBIx2rn7lyWOTmruaIpyHJNV3cilqWtivLKc81TmcE02hWIGPqaYxpg9xjE1G2c5zxSukNDCc00nFRPQZGxppNC2GfMpbHfmpbM4mWtLM1W9jpra4/dgVHctvORXn1/iO9R0RX38/ShjvpKCtqVYPKDHFWUBjHXrQ7Joi99Cxblc5q19sEagVjJNyNYNpWGPM0neqbOQ+eauMNwRIs20j3qcsSvvWLiuUu19hEjZmqYWrZ6Gkm+hadnYlitz6VYig+bnmiclYFuX7e3GMkYqcwK3Fczi/iYnuTQwlD04qwjdq5lO8xuKLcWeDmrcYAFa3v7pjISSMnmqsmVNZyulZDi7jFlO7kUlxNuU4NaRVlqVIpRu5erTRmVMEVsp6WFJajY9PTOcU6S229K5at3L+v8hX7lbYwYg06LBbHpVNM0TuaFvHgZ4qyYlK06VJ7s56r10KlzHtzgVnyFs4rePmEdtRgBBqdfu570Xe5TS3EYkDOeaiEnPNWpPqCGyENzULIT0puKjqOLsPijyOabLbpu5Ga55Pm3K8kH2RAM4pkZMTcVhFa3Lje2o+S/kQc5xUC3xc/MfwrtWqRNk9iRzG4yTUPC98U7RSJV9mMdh65ohHPWs5SstC1HQlJYd800xORkmknrYG7EYj5pcYNZ1VbRGsXcDuPelWNs5OcCsm3HRF3LttKsSZPWla4WTritLoh03e4iumck1PAyl84pTaTIlBmkkylMHtVWeZS+OprVKLu0ZJNMjZwO4qNHZz0pqyVx20J13AZoN55Y5NawkrktFKS+3kjPFVJG3k4FVLleiKUeUEBB9KtxjKisOWzuOQ+QBV5rH1C4Y5C1Sir3YU9XqZ0Ur+Zgir5m2xc8U5tNqx0OxVkuTng1UkmLvtoUbvUcUX7eILGDUNyWY4WsYS1Zp5iRF16mrkE5UVrzKwNJoWdzIKrLwcms1NN3M+XQR5BWfeE5zWsFrczasyoo8xuajucR5Gea6aauKSuUzLz1oeY461rydyJEDtuGaIIzK4HPJqtkc8jpdPshFEDitG2gR2ryazbkcFW97odqFsEiyKyEPznmlRuroyjI0rScx4yTWnHL5i8mm6l9H/X4nRfS48tjinRnPU81PNqFrmnptt5jgmu10mFY1XjFbYfe9iZo37KPcQa00YIlepQTRhNmXqy+apFcF4gtnEhPauHNYuUCEzmpQY2zUtteFWxXzrjzRGjYtTuTJqOecK1EYJxsIzrvL5bNZF4jYPNXR3szWm1c5++wHOSaphWdsKDXt0/hPoqOsUb2iaUXUMy5Jrol07ag+WvOr1L1LHTKPKgS2KH7tMmgB6gV2RXXsYN3KU0OG4FROoQ5IrSMVuZvsSRnI5FSxSyRuCpP4VjOVrocF3Ok0fUJAoJJ5rqdOvyxGT1qoSa1X9ficVemdLYXIKgsauNIrDqK9OLU0ee4u4zcq00kMawm4pWC3UtWqH61p2s7RnGaIask2bS78wDPGKmmkyuDW7TasyLEKPlqtxSKACcVmpW0YS2sPaVG7jFVZZI1NKdWNuUmzGiSM9x+dQ7kdiQaxSTGYPjBvLs2we1fN/j1zPqMg/unFXOUk7I6cNG+pyDKFPQ1NBB5gyeaUnpdHdboWBa7VpWjCkYrmctTtp09LjSuOlWrUMWolax10kkbFoG9K1bdCSCa4HOzNHKxq24JxgVpW0ROCamUlZnJUmjVtF24Nacdz5Qz+ddNLEOnHU4qmpNFqyowJBIHvV5dVidcgEmu+GMi47HBUgytNdLI3FQ+YCeK8/EyjJszcbImhl2nNa1lqBhwR/OujA1lGNjnlG5e/t3g/NThqwwCTXqQrXd7kuJHJfifI5GTxVO8sEuVywFE3zotXiUYNJjil3ADOfSugtYESAMVHSogrX0Bu70MTW2RiQBk+1ULSzViCe9eZX5ZVTsi2ompAixsCa27OdQnLZGK6aMktDJp3IdSEbp9K5G/jwxOO9eTnEVyK57eAlpZlINg88VNE27oa8zCSvoz0Kq6lyFytXYp9wAr0qTs+VHBiIaFqOPcMmmS2454rqqUbps8VuzIfIxTHhyK8yNJJtGkm2V5IBk8UtlmKYfWuujSe5F76M7PRromJRk4GK6KzmJGM9Oa9+iuaxyzRc8zb8xNYuq6iF3KDz2r0EjBao5m6uQ+Dk9PWs6ab5uprVKysWloV3k3dM1A8nrSfYuxWnk71SdiTzUXFYYeuc0wsBTTY1cjLZphPvQ1d6j6jCT1zUbGlfULidaaetOL7h1PmPoalt8CQfWtdmaxeqZ0Fo2VHepZE6kniuCvrK53rRFKVju4p0WSeazbUVYq9h5Yq1SiRnXJJFCXcLaDGnKn7xqVHMh3E1qmrXFeyJmufLFVnnLHPepa00GkIrktknFadgTIOa5Z9mbp3NSC2Xbmp2RQvSsOtg1bIicGpoRznOKL2WpaVtS/FMqqBkE1LEdzdah23M3FotACmEYOQa52kndDuTwv6k1ZSbB61uo31MZMWS6wD3qq8pbk0qi2Q46ELShTweaaSWrKq2jRLuOiTBzirAcL1rOm20TIBOB3pXuBt5rZrUOUpyPvbinwLg9KTaelxvQ0bccCpHJWt1LsYPcimcEcis2fhiTxWbepUUQeYM9acsw9apXW5XKSbgw61FIPmrSOomrMiOd2O1WIY1ZeTkmqkm7A9rg8QU8Gm+UCwNcc7xbRomOZQBzUDlc9BWcFrdhqyrcyArxWbIzK+RmvQUlypIcVYVLllPJNP+07+c1Dempokx0eX/GrUEe05NYTk0rD8i9HAHXP9KZcRbR61MG07mbK2AOTSnBNW027sE2ISFOetDTcYxUyhbUtX6kJnOcCkDuOay5nc3FWRs9eav2ZbIOaKlnoErWNSNSy+lRTwYOTj61LvHVHG2rkQiDH1pTiOqVRtlOPQd56haoXr7uldEL7tktNMjtIBI3NTz26xDNJXk20wk3cpgkvnPFTxzKg5NNT1sW1fYiub0EEA8VR/wBaxJ5q7rqCg0hr2+GyBUF0zAUopX0C92QBWccCmLA3m/MKHK2xtBamlsxH1/Co1XJyRWEN9TS4jDFJuxVyegmyWOXdwRxUc6YORURTvZCl5FORiGqOfG3nrXVZpJGLKRkKE1SuZck5NdNPRiexVJ3UjZNdDlczk9LkeWLYroNB0oSkM3NZ1naOhzVZJK6Op+w4hwF6CqZV7Z8npXjOT5tTz730Ir2+3ptJrNiH7ytoaXYlE0Y49y1NC7Rms29dDWL6F+I+YM1Iq80rp6otm5o5G4ZPNdfph4BzW9GT2RM9rHQ2TLtGasySfLxXtU46HLLVmXdzAkjrXNa7GrIT3rmxjfs3YleZxV5/rDVRG2SAk96+WjtqaLc3rO5VoetVLlzvzk05RdNKwmtSNpNy81m3xARjSp35i6WjOduk8yTGDWjo+i+cQWHJr06lTkpn0WDTe52Gn6ULdBgYNXGtRjHevLvKUkztmRvZEfMBVae3wDmvUh8N/wCvyPPm9TJul2tjFUp2x3reD9wlsZHKSeTV20O5uTWNSSL6G/p6jaK2LS4EbDJrnU5J2RnONzestQwAM8fWtaG63rycn616lKpfqedUhYf53vTopstROyepHLoaVo/41pwLuNVHbQyaNO1jKrn1pt07JnOT+NbSlaJCepViuX3VYF0wXoa55ydinFEFxqjxjk4+tY13r7xnOSfTmuOfMzSNO5QbxYVfDuR+PNXLLxXFtwXH1zRhZtzdyZ0rbGV4u8SwPAwDg8eteDeKLlbm9lZeeetdVWak7o68JTstTAEDSOM9Kv29qR2rGVSysdajZlgxYHIqGSH0FYJu+p2c2mhHFbsz8A/jWpY2Pzc061SysaqWmhsW1mAOladtZg44rjk7Myq1LGpbWuMVo28IXFZz2RwSmy9AACKnJBpzXMkZuRGYQWzU0a46cUqM5RdmYTvuShcHJpVHNFSbbuTe6uSrxyKkE20V0YeavuYyjrqNa6I706K5Zj1I/Gu9SsK2lzStXDAEmpZ7oIuSeK6YSajqZtFP+1Y1fduxVs6/GIdoI/OmsRBqzYcuuhkzTC4lznOa0LOH5cAda4pR99yidPNdWLa2bMMjNPSCZDnnH1rSMWnqEZdBLhnVCWPSud1BwXJ4zXHmMeaOp6GEnZmXKTvzmp7ZsCvCw8ZKrZnt1H7qLqk8c1LFIVYGvUScalzlnHmiacF0pO3tVlQHXNerCSkjxK1NxkRSR4bk9aTysivHkrVtSHsMa13ZOKha0KyDj3r2aNJOGhg3Zmxo8vkkKTwK6SyvU8vJIBrvoRsYz7El1qqCMjdzXMahf73J3e3Fd0XfQzjHW5mTTZbiqsj85q5N7FWImbFV5pKOpRTmck1C7e/NS/IaI91NdhRqw66DDyKjLc0k7sL6DGOaY3FC7jQmeKOvWh9xHzGc54FPhB3jmtF1ZpF6o6KwBMQOKkuAx4rklF8zZ3KRCtozAk0gHlmuWd3JovdXGSS4bNPWYbcmtYRUS7e6itLMN2av2WJUBxVP3YbkNaOwXikGqqFi4zWUXZIcFoX4LIynOM1pWlsU7VjVavY0jpuX43IOKnHzLnFZN323NLdSNlDGnL8g96xleKsupotdCN5mQ5BxU1neNnk1q17txtJo1Yp965p/U5zWPLcwk7FqBgFyaJplA96tRa6mXLd6FZpmY4FOCu655olJdSrWRWdJQ+exqxEhK89awqvmV0W2t0ShTTZDgdauEUQysJWDc5NLlnPeiq7LQ003J47c96nVNtc8Y6pkyZKku2lNyM810ppXRnZPUjlnB71n3c6gHnNTy6lQiZ5nJbjinqT1qVU5dzfksTIWzxzUmSOvWrp1LozqR0ARFuaUEw8Dv3rRz10MVroK7tjNV/PcNzWdot3NYpDmnLCoH3ZrO61RaXcgcM3WoJcIvI5rSLUncbtsipu3NyaesXNOcrOxqloX7SEnrV1YSMVzym7iciwrCNetV7iXOSTWlnsZlFpDnNNM5JrW9irXHCf1pS2fxpp9Q5R8VvuPNPa3A6Vhe7vYpT6DkthwasRny8VLheVxJ82haS82L1GaYZmnbHNNWaJ5LO5ahhITJqC7BUcVnypCvqZrzkVE7s/OK3jP3bF8pNalkOc81LMC69fxqPaOKYnEpPujJqtJK27GauLurjiiJmLtjNTwoSeM81fLzIt7Fr7KVGWrNvVXfipfNBmW7HW4Cr0FKQGfJFTLa/c35eo/b2Bp3l4Xioc+VibI3hIFMW3Y84qoSk9WK6tckVVQc1FNICDzWsfiIM+cs7ZFQMSAS1dGgm9NCjOTng1SlGOtdMe5nzdCLJJp+Nw61ehMtRLeBpJhjpXaaJAYoxmufE1FGJx1Wtkbsbgrg1DfWoePI61wVIpxUkjilpqYktixJJGarm2KvShIm+ppWkJK0+aLb9aibLW5NZuBwavpFu5qr6Guti3ZyNE4NdTpV+uwc81rQk+azIkrq5vWeoDHLZqea/8Al617dOcVEwa1Mu5vRyd1YWraiHQjPT3rzsbWtCwuS7OUvZgznmqEsmDwa8CmilpuS2mpGNgrGrcl2svOa6Zq8LP+vxG49SvLdhR1qhdXBk4zWcIamtGOoWGmNcSAkZya7fQdEQRZIHaqcueoon0dCLhC5tvZxxJ0FUxErzBR1JxT9hGM1/X6DlUbi7l+bTAkBb0Ga5vUZBGxHpXoVMP7K3mefCTk2YV9ICetZNxIScZqVJKLRpZjIyd2a0bDLMBXNKfc3itDorIEIKvxK2cmsJScnqNouwyunOelaFtqjIDuaqjiJJpXOedJSRbh1Ay8bq07Fd5HNdsZ89n/AF+ZxVY8iNq3XZyTitC2uVQgEjNdfPGLOJps1Le8jIHzCkmcTDP6VcpqSuSlZi2losh5FXjpsbLyTkDrSjCLjqEpNMpXmloFJPT1ri/EEXkkkURgtzanK5wGr3MyyMVY8e9Zn9vXcAOZCAK5PZrneh1KMZaMxNV8RXNzuUyNzxzWDKjTElhnNZTtHRM76dNRWgkVttateztUMecZpR9+9x1LpXQXFp3HFQG0Ldayq2i9P6/AcJaCx22w5NXrUBcHFYzd3dFuVka1qNwBrRgCqMmocW7s5qk+hdt5AT1q/A3HWhxStc527FhWxUsZz1qObl06GbJlFPztFZubbdjIFYk808E1lWl7tiojw2OtRTXASqws+jIqKxCs2/mpFZgQa9dS2MWXba7YHk4FOubrchXcCcV1QneNmTJdTFnkIfOeaQTvj71efdKZNyWC4MZHPvW7pGoAuuWP0rahVU5coKWh1umrHPEOhJ6CrMlkignH5V60YDU77GJrEexW5xk/nXH3hJkOa83Hp2Vj0cPNIoS/u1z+NFvcBTn1r5ypJUqy5uh9BSXtIXNOGQMPrVlIw1dsZe095DcXYuW8IUbs5/GrSzbOM16FCfLHU8zE0hS+9s5qRRuWuGb5qlzzJaDwNtRswJ9697CaxOaS1HRts5qddRZOhGa9JLTQhor3OpO6kFjn1qhJclj1reMUkJoiaXIOTUDOfWqj5glqRySYXrVOWY5qZPUCu75qMt70bjGs3oaaWFEdrA0RluMU0txzSBjWNNJz1pMVrMQnFHvQ+4NM+Z35PFEaEkfWtE+VDirM7Hw5bCdApy1bd1ogWPcE5FEqXMubv/XmW5tGFfhYAV6GsmV9x4rilBxZ2Um7DTGTzmmuMD1qOd6M0vdFeRCWGK1dMJiT5jRN80bPqEFdFqUCboKmsNK86QMRmuepdKy3L2Rv22jhUzimXMBh9q5lv7wRqa2IIiA/Jq5HLGQa2jua2b1IiS3KipRCSvArGfxcqZqnZEMsJGc1FGm1xinTd42Jvdmra5ABJFT+b83FS29kZPVkiykDjk09SX60XutdxNFmG3Q4yatxwIBXJUhJtMQyaFDzgVXICnFVFKKFG+wpA25qtLkMc1a0iNIhHLdqkiChqtx0HezLayqF5qNphnioUdRWYNNkcioHnxzkVVtkgRTurtkXrVAzySt7VblZNnRSSZZt7dpOSKuJZ4GetcE029C5S7EiRBe1NkwOa1immkjKbGiXYMmopZyxrrcUzK2oCTIyTUE0opKK2KitSJLobsGplkDnmo5bo0aJFgDjJqnfQdhVU4qJlzO9iktqVOTVmGPLcjmoleXvWOlS925dt4yD04/nVs5Az2pNamMpalWe5AyDTYWEnWtpR1uPZXILsAE7aqpGztXNLmi7mlPzLUVuwPPel8k7qKU7suWhchTFSvGcda05Xc53oxhOwZqJ5GY8UpqyNKe4qZA5qeCYIea54u6NJK5Ya8IHBqvMzz9zWkpWM3G2pQljKSfN+tSrtEYNJa2ZV+bYjMpV+Kd5xY8miTRo0hsi7hUP2ZTya1o03sZN22IpYwr8VPayCNx3rVQ5W+wNNovXD5izisO4iaSQn1NFe17omC1HJG0a96cI91ZN3Nb9QWNlPNSCTHFQk5bg2mOZgByab5oAreKUWZ9CncTnJ21Fkkc1T01K6DCBVK7lAGM1cE21oZme788mq0/eu6Ks9TGXcg465p6HewFUyZT01NnTLAZDGultSsSAdK83ESu7M86cryLAmyeDUyzblwamLvHlYpIbJACMism7j2S5rlS5Zcpjf3i3ZSjbg1NNhhVVbPYt6Mqhykla1lMrrVQfulplxUyMirltM0fGTVRfYDVs79hyTV1tRDKMtzXfTq6b/wBfeZuN9ihfXp2k5rmNR1Es5Ga8zGycpaAkZMk7MSTVdpC3FYRQrXImVs7u9TJcMFwa0eqHGN0QSzMxOelSWVsZ5AeTk037sbndhoXlodnoekBQGPU10cMYtoxisIxfxI9xvSxQvr8h+DUFteMsyuQTg1tGTlO7IqW5bGvc66htip6kdDXKancpKWORXo4rEKpZLocNKm0zAupMmqzQtIeB1ri5t2zo5ddSW3093PIP4Vt6VpJLg4zXNNuWiNrJK50EOnsijNWUg2r0FTa7JvzDvL9KXyielEYu7HYntw0fOTW5pdzkiuim7aNnJiKd4tm4k/y5Jpj3ZycE/nWtSq5WR5igT2V3I0v3ia3bZiVy3NddKL5bkTWpftTyDWiGGz71bp2Whi3qZ+rXQSH7wrgtbuPPkYen600rqxrTXU5TVLdDuJrkdUKnIAxXNW931Oqkrs5+aL5iTTVUMa43T0uelGWg9bfPJOKtwfux1oWmwXuSGUYPNRSSjqKmtFbkiREuavQQlscZrjcktCJzNO1j5ANXQrYqZTbW5zSZPbxturTgTaOaeqS5mQ2WByalXjkVk5dCWxN7Z6mpossMmrUls0RLuTomOaXByKwq6vQSkKwwM5rLvJ8NUUdJEOdybT183Bzn1q+Ywo5r36dGyuzBy1sUbvURDwKhh1DzeM5oc+iKtZXHMmRk1Ex21wTla7RFwR+etWIbl4XDA80qfuvmYjqvD+u4ChmxXStq0bRfM/JHUmvoaFROmr7hBO5y+va5D8yhxke9crLfCdyQeK4MVVjN8h6lKm0rjXbzFqq77Grw8woK3Oezgp/ZLtnejpmtK3u8HrnNLC1Fy/1/mds4amlDMGQMOahln2yZJr0+Zcpx1otk8N0G71cgfJ61zrSSZ4dWNm0TMeOtQOwU+9e3hWcj3IXmwDULzE969WF0RYjaQ5ySajMnvWq1EkRs+Kid+uaL6gytNMVHJ61VZiTknNKW4dBhYdKYxFTqxajS1Rkmm1Z3Q2NLY96aTkc00r6huJnFJmpYNaiHk80E55oXYLHzOafEMGteVlo7PwbJllyeM13ZtPNtsggkitltZCq6tHEeI9PkRycn/GudaN0zkVxYhcu5vTl2IzMAOTTfO3cZrJxurGyY5cbgTVtZQqZzU8hcJMsWcwbPOa3tInRXAJrCtHsbWudEtwjR8tWVqFwm4/NyK51TcjKzTMpbgtJhTUqs4NbSSgrdTeL11LVu3HzGrolVVAzmuR6yLlchmfJJzUQXnNbRipPQEWon2r9alL4Xg1jNNMViRZeBV2Hayj9amKursiS0Jo2KHIqdbnbwTTktyAaTcMmoHcM3BrFx1uxRvcN/FV5mJNDSNEyJIyWzzUrwsFzW2stiJPUgy4bBzzVuCLd71lKpZWZfQke2+XNUpIwCamnzXJUtSKWFW696SOyUDNKd2rM1g2kWIYije1Wj0pRSG3qQynb0qHIc0oq2pMtrjZkBGBVUrzXRzEq7QmSpwTTZADSctLjW5GlqWbPvVnyPLXNTTlrcuUtkKsp6AVDMNxOauFRSloRbURYQec04Iqnmm+yKvbYsROoxinTygJ1pJXYSRl3EhZ/WljMgPFaSkkWkrEhhZ+v41NBbbDkjNc9WXNoiouxM4AHTFQ53N7VnyqF7FJ3LCShB1oMpPFbQbM3G245RuHrTTEF+tTJO7sRF2Y0nHU0oAzkmslDsbJ2GtLipYZwR2rGcJMpaogvMM1RBCV5NbJckUJKyE8sg5NSCLIJxRTjfRjk9LkYwGxmluAI1JFdMU+a1zNsrxRmY8mnSr5BFaN30HHew+XUd0IRcc+tQRDc4J71FfW1jSMbFxbAyLk8CoJrbye9Zeztt/X4GTnrYhyTUTHZyavl10EyMzFjimuGxnJpynYaWo2OPe3NPmt/lyO1K93qDaRVkBVPes25Qk8HJrrpshvqVfII6moJ4/eurmu7kPUqSKQakt/lcU29Dnlax0mnyrsFaJckAg15dVWm3c4ZaMngkOBmrAfnOayUtQvck+0ccmqF2wc5qJq8lKJm1rchjJXoTVgTHGM5q5aqzKZHITnrVuynKYBNZXtsCZtWs4cVbTBq4yLJ0l2U7zyx45rTmaVkSl1I7nc0Zrnr61O8nmuave6kgRV+yMetAsyMk1lzW3BR1GvEBVeVdtVFm0YEKoZGAxmug0aw5HYnmis9LHqYKm9zp7WY2uM4qxNqSlDyKuDXJb+vzOyaW6MO8vx5pJPGaWPUEC4DVvRlCLvL+vxMJxctitdXryfdJNUHjlmbnNZ1Z3baNIpIemllzyDVuDSlHUfhXJzybt/X5Fqxow6cq9FGK1bC3EZ6Vcbc1mRJ6Gk0Q8uoxEac7xkEFoL5BJqxDYs4yRgVrRgpMptJEz28SRkMcmq1lfJFdeXuAPpmuqrCFOKaI5HODOnt2EkIKnPvSNbMTmpdO654nkN2bRNakxSc1vWcwZRyK6sNWVuRoxqbl+NyuCD3p7X/lrliOK1a3sc6MDV9T87cATXN3YYgs3WqormdzZaI5XXbzywR7VyVy5kc81yYhe9Y7KK0KktqWBx1NQmARferlbszrjduwySVE71BJfkcKabTep0xp33GpfF+9TJJ5nesakdNDKsuXVFmAbe9adox49K45dzhbuzVtgODWhEAcCsUrNoiTLUSDsKsoOK0sRckRTmp4hnrVpNpslslEWakRNorFt3sRceDS7wKyq1bIkZNcALjPWsHUJCJM9q0w0faPmRmnZk+magqHBNXrjUVZeD1r3IVEqdnuRJamPcM08h5qSzQxnJrz/bLmFJ9DSiUyClks2IzUSndbC2IhAVPIp/letXurCuTwhoWDKTxRfa9PBERknA9a6aNR04tm1BqU0mchqOt3FxOSxIA4xmks9Y2n5j+FcEZv2rk3v/Xc+ndFOmrGnFqkbr94Cknug/Q1vWjzU2pEUIuEyKC+2vnNa9peb8HOK8SlJwfK9v68z13G+rNW2vgBjdSzS7+QQa9pSTjc5akRtvc4kwTWvaXIbv8ArVRj7yPGxVKzbLofK9agm4r1sH2PKluVJGPU1Exx1r11uQMznvTGcCrQkQyTEHrVaW4IPc05JDtoVpJCTzTEJapafUAl+XFQO+KWzsiUhpbjNN3Z603LULiE+9NZqIXBCZOKTPPWk9xMMj1oBppFep817QOhpVbDc10XvqUjp/CE5WdRkD3716jp6boAOvFbwtYJ2KGtaOsq8jFcLq+nfZ92RiuPEwUnd/l/wBUqmuhzV2m1sg4p1pEWBJrC9lY64S5lqTOm2kOcdanmWrNLXRJaOVetO3vPKOcnNQ1zLTc1i2jTh1KSRepxUc0jMMnJzWMpcisUtxluCZQegrSUoQK56jT6mqgKgUHrU6vnv0rJrW5VrjGO6lQNVRk0yeo45A60+JyODzSb7lXurFhW3GrMDMvfisZS5SeXoXFOUzUe8hutaxV43MWh7TZXrVd5sHrScbCiCSk981LtPrUNJljlcIeanSZHWtIJW0MpRe5XuEUHNMiuDGetYyhrc1htZk73YK9apTT/ADcGrUdBcuoqHeenFThVUUpQ5rNBzWQiyqDyRSyXChM5qVS5W0ytSlNfL93P61ALoZ4NVaK0ZTi7XENw7HrU0a7hmspzinYqxHMMNmo1b5vWtIxurE20LULKpzUz7XHappw5dyZJuwxI1JJqK4VQOKSik3YpK7IN+B1qGaYjnNa0bWv1NGhYZHY+lSSt8vWhSs7Ey8isi7mrQgjXaBgU6j0sLoTrEuac6hVrikm2Vd2RUkkzkVCHwfStE+Z2Zqo2Q8SDHUmpUO8cVo7pEy2uyxGCOlRyvjilB3MbXZW3HdnNSBty9aHJOdje2lyJzhuualtwSM0m09BWCQHPNM37RjpSla1yoD4wDyeaeeAcVFF3lqE0VHjcvuGaiuFkyBgmute6Qmi3ZxlY8nHSql+5DYqLprQUfiKkSsW56VfhVUYFqKrk9GaS8i+blfKG01nXFzngnNXHaxhFEStntSOm/wCtTOUY9SrEQtMHJokHGKy5rPUtajliwuahlnYHbV8ybTEkmVp/nqpIMVvDXRENFSU4PNU5nGTk5rqitSJIqyMCaiLlTWyRzNamtpdyx4JrbS445rhxENTkqx10JVugKkW89W/CuNwaZCQ4XYfvTHcua0SS0YND41zU6xA1lVVmLlsK9uNuRUCkxNzWXUlGpZTlsYrWhcnvWi0kl/X5l27kzZp8ByaHe5VtCSQbhgc1RnttxrKrq0ikupF9kA7UySEBeBiuWsmmkaxjfUzp4wCapSRlmrWmzWELsvabpxlYHbXXaXpZjjUtUSblPQ9iCVOFhbxQrEZrPmJCkAmtNWZtmRcwyO/GafBbN3JzVXb0Zaa6F2K0HcVPHaL2Wole9mxN6lhbfA6UixuH9qdNWdgbsaVtCSgyDVuC2fzBirlZO6EmaQtmWPLEdKh+UMRV1tJWZpSjfYngWPeN2KsXN1HHF+FaUJRjHnbLdNto56/1N3LBTisqK6lS6Dkk8159fESnJWOzkioNHc6Del4Vz3roI2BAyM171BKVNHydeNpscY1PNRm8+zdD0pTpcr5kYb6MsQ+JYgMOeRVa78QJOdscmfoaaxCcfMlQSdip5vnHJNZ+rXiRIRkV204qMQe9jgdbuy7sS1c9JfRxN81ee17SbZ6FPRWIZtXjUcVmXOqmR+DUOC+f9eZ3UYa3ZA1yzck1E7nqTWdjr0EjlYPkmtG0mBYA1FRaaGOIhdaGjExc8Gr9puLc1wN2djynobNpnGTWjDQ0jJlyM+tWEPFJNbMhImQjFSoQO9JPlYmTqc04mpluzN7kbuRUTzYGSa4cS21Ylszr3UViBLNXN6h4gQZyxFe3l0F7O8jNK7KVtr29wVbFasWqNKMA1jiJunLmX9fiaSWmppWTGQZbrVwsFIrGNmrmK3LlnMBjNXTKGHvVqz3BxIdmWzS/KvWqvykiNcKorPvcTA4p1Ki5LRCm+WVzDuNKLuWrOv7NrVcg4rKlT91yf9fgfUYbE86SKENzKZNu4kfWtaF3dOTRLEOMdf6/E6q84wadhTuU4BqzDeNB1Jz0615sveep1UK/tFZFqPVyGHzn860rbUGkHBOPrXdCslGzf9febVIaXZajY9au2t2UxkmumnPmdjzMRT5omvbXe9Rg1JIw65r2MLa587VjZ2KkjYPNQk5yTXrRXUxtcaWxUMkmBWltSbFSWTcc5qFm5605PSw7EbmmhwtCd1cEiOVyTk1GSal6gIfemn1zRuS9BpOKbu5ot1KQpJJ9KTtSVrEsUUuMdaNgPmnBHJpAxHJrfdlK3U3fDE+LxVJ6169oDNLApz2rek+n9fmE0rI0L60EkRznnvXB+KbHOTgjFOrqiIPXQ4W8tz5hBpbWLaOa8uro3Y7qe4+ZScVCc1ErcqsabDkbYeauQqZRkAmr0TuaRva5owHYozxUgcu2M8VxVdZHRCKauSqu05zUqyH1wKxcbaDWoGXvk0q3RGMmokpdDRdieKcSHrV2BC3TJzQldEVI21JGgHfrQYsAYpyfQyTHRgh/SradOtYS1ZoyzG2BTJ+ASP0raMbHO3rqUmuCpxkj1zUUk+fSq6tM0jHW4sVx74rTgJkXNc8pKJUloDQsaRswrxUxqN7GbV1YpXN+B1z9arpeqx4Oa1lF2NYQ00HSXfHBqLzmY5zTg1sJruWIZyvalmumxz3q4tEuNys12/PNRNdSt3NTKp1NIpW1IGclsseTTo23HrWdRvdGnLpcswjnnJq3HnHNZyhfViewSR5B45pgjx061urKJk9xTkc00SHPWolotBxtuOMmASTUTMZKcVZBzDDGccikS0LtyDUzk4bFc10S+SIxwKj25PPNTCXVkrUkW3JGQKmjDIOlNtNNIS8yVXY96SVyQaIJWLRUkI7Gm4BHNRZ30NlqhVjGM1IlNuz1EWlI25JqpcSc0QdtTNK7IN5zTvM+Xio5by0OiysRNMS+MVbtN2Oc05T5SEiUjceaZLFtXIrK7S1K2Y2MHOKvwWyumScUou8yKuw2WJE4GKaY4yuSAa7FHozkMy51JI5Ci9jVeaVZea1VK1joirIjSXDYq26/ux1zilN8zshzVkRGcxRnJqqbgs2aqdopNiirlyIZTOKj3EPXNyc87l2Ay5OKYwwd2KpxTlYQ9JAc5NVpV3uSBxVWjuyeo02+etUp02MRzTpuzZm5a2KF1g54rJuGIY130kZzIGc4pFyzCugxk7mvpMB+9WgzHPWuCtL3jinuIHOc5NSAsaxbJbsODstWoH3dTUWBO5ZQ4OasRuD1pStJDkTZ3VHLbhucVzSjyyItYltf3ZGa2LNw7AGm+5VrmhNCBHuFRRkjvWcpuM2mXDUsxkMKjkT5s4rGVnI0tYiePPaq8kWRzRyNo1i7lC6jAJ4zVa2tWlm5GBms4uyZ2UIa3Z1ej6YEjBIFac9wLdMY5rppRcFzTR2SadkYs0rO5JqJwWHNNvuIZ9nyckVNDAAeRUpXY/QmMar9adHGx6A1aScrMW25NFHlsVdhtY15Yc+tXGN3cmTZqaXDC8qhmAAraawhKblIz2Ir1MLRhOnLuYym1IoXivDHWYJOc5rxMcpqouY9Gg1y3GtdBe9V7i+LKQXNcKnLWN9P68zre1zNlYSMQDUltagnca6cPRb1OavVsjc0ib7OQpPGa6i1u0dBk8172HfupHzuIfvNkrzbRkGsy/uDtJzitJq+5y+ZztzPcSTYjc9cVq6TYynmRi3PNYYeinU9CHPUsarciwiOCelcFrPiwhim4n8a7aza0NKSuznr/WlkjYlucVx+r6q6MdjH865qUPe0N5VeVmfHqbyN8zHmrcU2/qaupHl1R6NCfNG5OsuO5pfM45zmuVq71OyDuG7AyamgmKPnnFQ1cuUbxNnTp92MnrW/YoCAa8vERtLRnk1o2ZrW67RVyLPFZq8upytlpKmEuKfMrk2HLM2anhc980X5iWWVfApxfjrWcmzNoryTHBNZWo6oIFbLdK5JRlUmmTucbq3iJpnIDHA9DWBd3zzN1NfQUvdpplWQ6ylZG6muk0lzIy5JxXPX95NP+vxIkdVYoSgq7HbPK3SuSCbehjexp2ukyYB2nFX7fS3YjI/WulYebYX0uXP7E+XIBrJv7Uxe1OtR5ItEuV9jNeNyeM00RHHOa44RbZPMRTptUnFc1rW9txx9O9dilaNj28tmupjxQzK27bWnBOUXmuRq+56FZqS0JlkVuc1IY/MGRXNOm0rojD1HCaHw2bZ6ZrUso2Qjg1zwq++rs9udRSiasPSpCvevVTtqjmlsT287RnrWhFdhxnNerg6reh4eMpdUMkfJzURY5r6CPQ8uzuNZsAmqs8hx6VV9Lit1KrsaYScZNTJlEZamMwAquhNtRhf8aQmpQJDS2KY5wc0JNbB1G5J5PSjIzmmFhGNKDjnqahJoTsO3Z60ue9FrisfM5c5zRkkcV0vSQ27o0dClKXS84969i8LTb7dMcn+VaxV53RV1yHSsvmxdifSua8Q2BkDcda0qR6GCbueb6ramOY5GMGqKlR0rzaqakehTva4kje9MVQTmuVbXRs9tA8sM3FbWlwAQgGs670T6msLsttbKTwM0zZ5fSsGmpJM2V7CCTccZpxq5LW49mNZiM81C8hyKhXNI7k8BO8HPWt7T5hsAJ6Vb+FEVfeRYeTNAA9q5ZyS0MkiNpF3dRUkTYOc1NOV5cpclbQm88joaY1x8pya7UklYwaKEz5fg0qRFhk81jVnys2WxPFaHrWjbMUXBriqSUrpDWpPv3VBcLuBoobWM5OxiaihyarW8T7t3auxzSjY0hL3SY9etTwRbj0rNW3YS11LiW4xmq1yOeaUlcmLuQiP8aXyht5pKPcvoVLiMhuKIVZW5FEp62ZtDYuw8jJqZHxzmha6GUr2CW6A71At0Gfg1bikQldk5lBT61Go3NxQ7JCSsPeMqM5pYlGeazTursUttCVlBGeKQYFE3eOgo3GSkAZJqGJwz4zWFmmbJdTTt4VK9afJAuKI3u2TJWI1hAPXio7kAKTWkNyWzNlJDZp6Kzj1onLsdMXoPCkHDVYjtyRk1i6jcrIJMeYSwxmq8to2cmtZS0I5tbFaVCpwKdHnb0rJz1ubLYWNQZMsK0YwPL6irlSjKPMZMgkYrySKVQZRzmpdmkkW31HpDg81KXZRhayXuSuS9dyCQM/JJqCZ2RD81d0ZtmTRlyQeY5ZjQYMd6l1JR2NYKw6OLacnrVwEMmDTu0wmrlC9V92BnFSW2ns2Ce9FSbcUZp2L8lsIIaovy/tXPGbvzIuLvqKIx2NSOqlPWt1rsDuysE5NOUKpppc2jJkNkk25OazrlxkmtrJKyMHqZN5KCDg1lyA5OK7aKstSJMiIGOtFuhaQCtzJ+Z0tjb7YRSzptrzajXNY45rUSPk9asIPWoYt0DJmpIFIas1clFtWqVDmo5n0KuTxuQc1MpMnShpN67hbqTRWUzc7au2sEsbA4PFc05crt/mVGS6GoDI6YIphiK81z1JS3LjboPgVt3NWjHuWopJyRb8iNrfNVZ4zHnNb6rY0pauxnyx726CrNpahMHg5rn5W27Hp01Y27K58tQpPAp9ztm5ruvenYqSs7lJocMcimCMZ6Vm43KQ8xhVzzTYfmk204R+yga0uXksg2MsOa047OGO3AyB716mEwq5pOauc1So7pGTNOqTEAjg0pui44NcrajJpGyVrXJrW4kU/eNadtqdwvViw9DUwqTUk4uxE7XJp7/wA1OVx9ayLibbkgmjFydTVmtCdtDMuL8g/eqr9v38A8V5ShaWp3yqLlLNiPMbJNaqJtFerh4+4eTia2pYtgd9bVtvEec8elenQg7HlVampKl0d+NxouYjOtTUTvysybvqiK200I+SK01kjtoeo4FddCKhGzMuU4/wAY60ojb5h0rybVL0yylsnr3rCrV9/lDmsrmbcXTlT81Yl4WdvmNKD95mcpFVHKPmr9vOT3rWaT3PQwVToXUbODUiyZPWuKSPcjqhdxPel3HPWosbRZesbgq4BYCus0mfeo5zXFiYq2v9ficOLXU3Ieg5q7EcYrhSd2zzJFqPmn7MnOaclpdE3JY4vWrCLtHFRa2xDux27HU015felNrdEyZWuptkZPWuH8UaoYwVBwTU4ZOU1cmK1OTE+8kk9aWIebJgV7tVLcq/U3tM0jzADg5rodO00QEcV5Fao+exzyk27HS2EKkDNbFnGiMMgEk1VFpSuxSjfVHRadbq6g1eNt5bZ6ivcsnFMwRYYqYu3Suc1iNNxBNYYhKSaGjJMaA+tRsoPauFRilZEJFa4gyD71l3Nkr5461hUjZ36f15Hbh6ji7IrHTUxnbVK909gDsWseSz5kv6+49KnVblqZ5hnjPINWrOQ7gGzRK+zOxtbo6Kyt1lQHFWvs/lDJFcNVezndo7KFbmViaMgDtmnFvevSg+aOh027jt+BSR3BU9eK76EnFpHHWp3TRaS4Jx6e9O3gnrX0FCV1e54NWHKxkj4HJqlOcnOa6VsY21ISc0wnFSw6Ebv2FMJyaWoNDWzmkJ709CdxjN3ph9SaoFcUdOtNJqbjFBB5pcAnrRrchjh60h96Fq7iPmfGe9OH1rpkkimyWxkEdypOQM1674Kucxrz1ANawHrY7u2IaIemM1naval1Jxya1kroxejPOfEdhsZiF45rlnGxyOledVi7ndh5dGROSetNBYVyeSOtxsSW7/vRmugtJBtHFZ1IptGlOOnMWC2BnNU7mY9D1rJq78zVboImXbmntMoHvRytdQWpDLKW4ANEcJkIJzQ1ys1joi9BEOtaNsMGsJtoz8i0eFzULylcnPNZp3bYo9iAOd2Seau25+Xmptrcuew9n4JNQlw3Ga35uqMbDkgyd3WnkFfpXLJ8zbLS6j0mIqdZBwTUTptaoZKsxqC4uccEmtKUbbmTWpRcCTkikZQi8U9JSuWVwMvk1ftSO9NpuRUloWHkAHBqnOxbPNXbTUyitSMdOtPyCOadk1oaMVYkc5NOa2U4xXFVWpabHi3CrnNQ3DGMHFbUtXqQ5XdmZ1xOxbHOKS3Vs5rpbsrDirFxMnvU0fynOaySbeo5WHvPx606Ebue9DguhjayJ/LOMnpVaVyp9Kdrqw472K8hZ1wKW1gKtkk1jUubxdjWiJROtI82e9Z31sNq6EEvHNMlYOMCrVjJq5VaAlsmrECKqc4qXGLZottCOcgOCBUiyDYOeadJJtsT2HpJjmkml+U03vYhbmbK+Se9SQHjms5Ru/Q6ehJld2TUrSjbXRBLlMdRi/Oasw8HGay5bsGT+R3pRGACTU8q2ZEpXK1wc/dFV5IfMXFXGTjoh9ChLEYyeKiZ2CmnJt2RpFkSzEv7VZV8c5prfUt6gCZ3HGea2baMLGBjmsajWxnOKQ27h3KaofZi2cClTaW+wk7IiMRQ4OaSX5V4rek76sbepCjDPzUy4dVHWreq90mRTeX3qhdyHnOa6LN2Mn3Mu4fOetVXHpXZFaGMiFhjmprFd04zV9LmTOrtI8xim3UJrzqusznnoV0jIarKgkVL2I6XDHNTx4/Gs5rqTqSBGJGATVq3tpZG4Q/lWOtrlaGrY+H57ggsCoPtXSad4XjjALL9a1o0XJ3mZyqWVkaP9jRKOgqGSwjVicVVajF62/r7iYNsiaBR0FRPbZbivOrw5VZG8WSxWmOcVOlsTSpw00K5iwunFlyRWZq1uIl966nR9y7/AK/A2w8m5owt37zj1q7Ax4zXDpH3We6oaFuLOcirSBnGACa6VqvdIlYkksJRHuI61UA2thuKqrCUGubqKDT2HSuirx1qj55STKgn6VajGMrx3G1pqSS6o4GACtSJq8jx7SxP41q68ubdr+vUlwjYgM+5s+tW7Zdw5J5qEr69zObZfgj5FaECKBzWkIK+pyTkFwgxxgCsa/fbkBqwrxs7GlKTMWWJnfvTlgKjkVh7O+pOJxLWiLds+wjGavre4XnjFdNGryaPb+vM8+Vbm3CLUwkmecVoR6/hcZrup4tw0MHdk1hePcyZ5AreiJCDikpuTvIuK0sNluVjUknFYGsa/wCQjfN+tdU6nLG4pI8+8Q6210SdxANcfdShpCc1xQk5y5jmlduxWlf5Tk1lXcgJ4Jrtgle4alNm5qe3m28Gt5R0N8PLlkaMEoI5zVlGHbpXFUi+p9DTnzIfmkLYrFLU6E9CS2fDjOa67QpMqMmsMSrRZzYn4TpbZjtFXYmHWvLi7t3Z5Ui3Cc1YjGTQokMmWn7jQ2lqQyOSQioHlrlnJbiKl/KRCea8v8Y3xWVzvPB/Ou/ARjOaZPPY5mDVixxnrXRaGrTSK5HFe7i4xjF/1+pHNod5pqhIx9BWlG+SK+UqyTkRY1rJsAc1p28h3DmtHJdzRLS50uizquAxOK1JJA4Ar3aVXlpKLZytakZyFJPQVz2qvukPrWM5psnW5QMRIyTUMrhOprkb5fiC92V2uA2RUSosj84rPmVSdjROxrWWiJMowBmpJvDi5wRmvZhgYch0wqmbd+H0TnaM/SsuXR1Vs4/SvPxOGWnQ641rl7TYvIIB9a1XtlmjyBzXn1qMZU/e1Z0UKvLO5lyxvDJ6CkE3vWNByhaDPolaUeYUSbuDSEhT1r1oNLc5pL3rDvtO33qaOff7Yr0sHX97lZ5eLoNLmHSScVVfOSc1610eTYhJxTC3FUrMLdCJiQaTJpebE9Rpc9zTS3PWiyb0E9tBD0yKYec5pXDzDO0U0HJ5qlZ6hbqPC4pR1oWpIuQKQmizQrXPmYMw9Kdzt561u9XcrSwsTFZAfevS/A14SI8tnt9OaqDS0ZTtynqGmncg5zU17bmSM8Z4rqTTZzS3OA8UWL4ZuuK4G8gkEh471w4iEb2OyluQrC2cmh4wK4U1qdTk2RwqfMyK3LNCF3VjLTU6oytEsFj25qlPnf2/OhJX5gckyNZGBxSxsWJNKb0Li76IsJGzAVdgtsEGsJSvoi/Iuxw4GasRCsq2m4JdRXJwagbk5qfhXqTu9AQd6nSQgdalK6bKepFLMWPWmxbiwqZycVdFQStqX4A3vU5ty4zWDk7JkyaiQsnlnGMU5Tu4zXRvqQtrkynA5qvNGXbOaLXV0SlZldkKmkKkjms1JI1jawqxDHNK0nldK0pq8mSxjXJPU1FuLGqqT6IIrUkWJs5qQpsGSayg5XswmQGc7+tXYH3LmirFD6CvNg461BInmk81UI2ldiZWks8HPWmhdh4rWSb1QkyRGp5OFp9LBLci3lTkmpY7kpQpXBq5N9rO3NQljI3NTKdrgojtpHarEC/SuV8s7XZotiSRyBgHimYJ5pySvoNS0GM5FMEzDijUEkyTzDjmgSkcGnFJMHohsjBvekjJB68Uoys9B290cZioqF5JJD7Vo7LUlJLUEh3ckVJsCisle+pV+gx+TUkMZYYPenzcug2tCRk8sVJD1BzUOcov1M+hcEgK8VHI7c1Tmr2ISGqA45602W2YLkcVLnyrQWzM+5jbJ4qmRkkcVdLVmiIzAQc4pdjYq38d0aQehb06I7smtU8L15rGpFJ36ibVyNssPm6U0KBwBThqtTMguI9qkms2V2yQeKum+XYSZWlcr3qrLPuySa64Q6icrkSB5c4ziqd6WXOTzVprnM5voZU8hBJzzVQyuWzXdBaWMW+4h3Ma1dHsHeQMaVaahESjzPQ6eCEqBUjQb+orynO613MaseVkX2QZyRQ0O0cUrO2pit2ILdmPStfR9AN2ctmpinJ2JbsdJbeEkQAsvXvWna6JDFj5F4reFGKepi5s0re3jj6AVaBwO1dKtyWRGvUa53VUukYdBXPOWli09SqIWY8irMNsvUiuSUeZ+9sa37E/2cE1PbWO45IqaNL3guWpo0t4Sx44rj9evAzkZ7V24h2p2OjCfGYSSbn645q/AQSOa+fd3I+lirrQ0bVQxGelbkLwQwg7U4HWvWy6VO75jlrxeyKV3rEZUqOvt0rIluN75rHGVVUasVTp8qGjMjAGr9vYjZkgCunCU09Z6nPiKhkaowilKgdDUUBLgVnVglNsal7ty3HDuIq/brtHTNHJ5mEpFyOQjmphckDNaRet0zCWojXe4cms28jLMTnisK8k9Ua09yuIQpp7qrDiuanLR3ObFQb1BIgBTJOOBXQ4rk908xt3sPtrGS4PQ81s2Hh9iw3DNVSo1JvQ0ijds9KS2QcDippp0jU5YcV6ypqKsaI5/WNWWNSQ2Me9cTrGqGXdlv1rmqz5nYzqOxympXBYnmsiRiWqKKsYxZVuZCARWdKW3V30xvsROhxnFMG7PFbphHTUtwswIznAq7FOcjqKxmrntYWomrMn83I6U3zCDknmudR1PQi9LE1swLgn1rrNCc4ABrlxCtoZV3odTbkhRk1diOa8hrV2PMky1ETmrcTEiqV07GTJ14FDPUuyViSCRjULHnmvPrS95JiZn6tP5cLc15T4rBmmYDrnmvZy7lTvYwm7Ix9N0tpZga7vRNPMKLmuzMa+liYanS23yrjNXrcEuK+ZnNuRdjaso/lzmtG2Q5BOarqnc16G1YMwAJzgVs277gOte9B+4c09NSxMNsVcrqrYmOTms6k2o2RjsU/tOFxVO4LTHiuTFzc1yxBKxALeQc0uDGR60qUJxtzC5rs3tE1FUKq7DjHWtia7jK9ia+nw9RTgjaFzLvHV8n1rLmTvXHi43e51U7rchII5FaOmkuMGvNnC6ZvJ+7cl1CwBTeRziudukMMhya5J0uWVz2svxPPCzIo7nacE8mnNNu6da6uZ7o6m9RQWPWpoxIecGtYykp8yMarjKOpKxOOahZz3r6KjLnSPBrRtLQYeR9aYRit9nYwI2waYTgUPsCvYZ1ppyDxTS6CbA5700k81JKYhPHNCjNUkO5IF4o5xTdtxMTpQSKV2LzPmcKetAJzz0rqugQuRu612nga92TKpNOOm4JNtnsWiTb415B461sFSyeua6dLaGUkjn/EGk+bGxI6g1wOqaIUJO3FceNp3jdf1+BvRlbqYF1b+Sx4qi8bMeK8+MU1qd0X1JraxYOGatWNNidazk0jZSurEckm04FRMu85pRslqCeg0pzT4UHQ1M1FmkHbYvQxjHNWI/l71zqKSuap9yzGCy08EpznFDSm7hzEU0xx160xJAetZ1ErFRWmgrSY6GkEjMeDSpvSzQWHLEWPXmrtta4PNZVlrYfMWchamEoCZzRCKejIkivMwduaiZ2Q8cVbS2FFdGPjcu3U8VMUO0mpUlaxLKj5JNBU7ayduhqhnzCo584J71UJ6jkimznPWpY36VTtK7RpbQtRzgYpZzvXOaFLVGM4lMId+Qasxy+WOtVNxbsw6WEMu49ab5xBquXXcETQyCQYPWorjaDxxVS3sTsyuJCDipC5x1oqFKOgKpfrTvLasdb2Q1a1hQDnoaljUkdKU49Wx6D2UgU6PcBnNZqFtSbksS7jzyanMHycVMpEy3K0kRYnFLHakda1ptWSZTmkhJ49vINV0VnbGTVtrZjjK6uyY25AzQUKDnisZWWxpe4wn1qWJVwTS5lciUdLiM+08VHlnP1q5xXLcmNtx6RHPJq3bxDA5yax0tYp3sFxEWOaiVDyAamt0RMGSR7kbmpm+aoT5rFS3ERCDkipJZSVxiqTSeplJXKNwMA8day84nO4da3ppRSbKir3sWGUOowKTyQaJTSky4ppEqN5WMVahd5utZVJhZWuywISF6UxIPmJNJXSMrkV7gLgYNZMyjJNbQS1YkrmddAmqEzYPWu6Mk1YVjX0W0E8ZJHasnxBB5MhPStIxV1I5pN8+pzdwSTmoVzurvjytCaL9jZ+fIK6rTtP8tRxXnYuetjoo+6rs0I7c55qf7NxmuSCT3OTESTloNNtu7VImmF+elEpWVjmbaLMOkYxxmtzRo/szcipjPkkpIh66G79o3DNCyFjXU563ItYnXI5qZG3DmkqttxNdRwQg0SxBhyKcY82hJD5WGqVYTxxXLJW0ZpEswW+41oR2/ljJ4rooRu7ikYfiG+8tSA3A7VxN4zysWJzWWKldtdjsoPl1KBLI3FWLe5KkbmrypR5bn0WHmpI04bzAyDSzakSpG8/nRSmtv6/MucTOnv8AB+8PzqS0lM7Z3ZroiuZ6Gc3aJq2dsWcEitcYROa9GguRXkeXWldmDqkAklLA1BBHt4rmnLmnqilL3S5EAPrVhXI61ajcxY9ZT3pTKcdazvyxFYiluPLGaqf2gHfGa5XLmTVzeKshzybhnNMj3McKc1zxUkRU1WpdgtJHHOau2+lbuSPzrtpXdlY8+cNTUtrMRYOK1IZFjAJ7V6mHkoEqOg6e5Vk4YfnXPavePEGIJ46VvVldXWwbaM4zVdVaRmBbiuW1O/JY4avPVmznqayMeS583PNQSHjrXSo2ZMUVZAWGSTVZ4yea6INF2uLDaNO+0VuWPhfzUBK5qnIpQ1LNz4b8pcgVjXdm9u59qyUlc6qT5GVllYHkmpEcnmtLK56kJ6EschVgfSun0G7J2knFc2IUeXm/r8x1dUdfay7lHNXoZOleJJpN2PNki5C5q5DJxWfPd3M2icSZFMZ/U0p2tqZ+RE8mTVeaYIDzXJOzauIwNau8owGa4HWUMsx9zXr4RWadzKb0sT6NZHcCRXWWaBEFLHSTdmTG6RoRPjFaNgdzAk9K8nkV7lo6GyUbBWrZx7sD3qqfK6lkN7XNi2tsLxV2D5Djk17MLpGDehPcz7YuvNchqc++ZsYxU4hpQuzFa7lMKW5p6Jhua4ox95NlSJHA2mqskRYmt6lrqxCVh0FvMG+TOfSriyTxj5ia76HPCzZ102nEgluXIxmoxOSeTUV6vLubqw4EPV/S/llFclOd5cty206bNuWEPDg1xutgRs3YitK3LyXTNctnadjnDd5lwGOfetWwjMoGT1oo2m9D2sRPlRrQWakZ5q9DZqR0r1KOHhe7PHqV5XHnS93IqCXRiTkDtXXTpKLumc8qvMUrqxaGs6TIbBrpTV9Sd9CNmyaaTzTs9wsMzzSHPXNHMkIacmkyfSqTQrAM5pwFDWtyWtR2aMnHapv0C3USm5FU7dAPm4D3oMeelbSRKvuNaP2rc8LXPk3ac9xQrsuG57N4Zut0CAHIx+ddbC2VHSuumrpGFRBcWwkXnFctrej5UsF5x0pzhzxcWTB2ZwOtaa0TnK1nW9urHBrxcRScL2PRpS5tEXfswReBUEuVGB1rgUm2bxZB5e48k5pSmOc1o5u9raGkSSGDzOTUnkBeaykne6NFe9h+/ZinJLucHOKb0RSvc0bVsjg5p0w4yTURummD0ehUkXOTTI92amck7miuPKbjzU0MJHPSudVGnZmjZYRQHBNXFcEUP33Yz13GF+fWnElhVRVpDYGJjTWgalJ23JvYdBC270FWjASmKzcnbREvQiNkByabJCAO9VLa5UZPcheLHSopoQV96S0Xqap9SGOyDNyKkmtEjGRUxuldCcncrN8vSjcSOWrSjfW5VTYjZz2oG4jvQ3bcnS1xcH1prbvWr9owQJKVPWjzN7YJrRsm3UcE75qQAVlNOw73RJGnNWYYDJzisnJqVyZbD5LbYtMXA6mreq2Ii73JdoxnGaYRz0pqLQ46k0CDOWqdm4xUqD1E3qNSI5+tSiLA5qVJLQJMrzwb6jih8tua15brUSl0LI246VBMu7sKiUY3KT1KkkfPWnR5Udazhfn1RrzaWBvm5pY0JOaqq+go9ycJnkmpoiAQK54aWY5MmcArmqb5Uk1pUSfvGcASTPBOTVqEZqGknoW0WkRelEsKKOaztqYyTuZt6g61mzBc5710fYNIJsImOOalSs115jdaIXZlhWlZldo6VE4uWpEtid1Ofao53VIySea3hFt2ZhLyMmaQsSM1Sl5B5q5PlKRRuB1zWXcsFbkV0UXdXJk9dDqfDKq9sW74rC8YwsrNjtXWlsjgk7VDktjPSx258z5uBXWpJGy3Or8P2CSIGIro4bYRivJrNuTuE5vYnWLBpTHmsYs5ZO5ZtbYMecVpRWK4GKStJmciwloFHSnCMKfSnypamWtydHIHBqxHJimpNvUJE4k4zT0mA6mteW7EkTpODzUhlGOtdEJLYSiFvCZW61bFttNc86Ta5kWmXbS2zyRTdVuRawnnk11U4JQ5iN5WOG1a8NxIc9M1kzEdq89vmbZ1J2KU+B3qk83ltkGolTUkd+FrNOxLHf7Rgk1VudXwTk4+prnjRbdj1HWUig2pb2+9zXQeHrpXxlhXbThql/X5mFad46HX2zKFBpLu4G0gGuqVkjy1e5lyPuY85powK47e9qbtkitipA/vW0bGbFMmOtNM2DXLN2i0xqNyG4kMiEZrPSF/O79a81P2ctDZaI0BbOyDmr2n2YJHH1rqpXbsYTlc3YLUAZIFWkQr0roho3c52xWcR9TTJLgKnB5Nd9F6NEPuUpbpk5LHH1rB1zUtiHLdadSTUbNkTktzhdW1DLn5s1zl9cM+Tk1lRSepyPfUoCZgakEpI61220KiNzk80m0k+1TsbwjqaWk2oLg12WnhEjHNKUmkb8ltRupSqIz0rk9WZSelYxjqrIvlujFkHzetaOn2PmqMjNdFR2RTm4ItXOmFF4GKm0kNFIASeDXI3zRsXCtzxOx0+Y7Bk4rUilGOteVWWtv6/MxluWorhcjmrsUue9ZKLi+5LjoThzTJJMd6mq9NTK5Xa4Az81Z1/ehQefyrmjG61NYwuYV1N5xOScVmTWSyvnFd1OTjqc1XR2LNtaCPoKvx5UZJrOrPmd2THQmV2wM1p6bMN4yRn61zSfY0udTp75Qc1r2DHeKig0qicnr/AF5hLWJ0Vqu5QTU2zBzXvt+7ocj0KOqyukdcrK7PL9a4sRzStFk3LEcY2ilEZJo9m1YlSJNgUc4qJ2TdWtlzWYtS7p0qFwue+TWhPaxvHkDPevYi4zpXRcG1oZF7brHnj8qybiTYa8HHSsmjrg77j7RzK2K39MtDkE9c1lhIuUk7jlK2hsXC7YBzyBXE+JYS3mHJx0Ndlakp3LwE7VDjYMmf5jjmun0tflFTgFoe1jGb9pCSBxWhHAcA9q+hpxsrniTd2Wo4vrVpIFIya6OVGD1MnXLdVQsO1cfN8jkelNt3sbQu42ISTnOaYzGq5imhOcZzSM3epsrisNzk0AnNXoIcM96cKUtNiW7jsUhPWpvqIY3NAIxTbdtA1Pm0Nz1o3D15rsa1uD2sL1HWrOlzGG6U57046scLX1PYPBt6HhRSeeK9AsW3IOee9dEHpoRNa3Lm3dxmqd7ZiRTWqdjDZ3OQ8QaIHQ4XFcPfWjWMvtXJi4prbc6aE+V2YR3HmjrTZEBrw6lO10ejzJIrPnpRjPXpWTlZJGi0Q6OTyzjNPM2RjOarlfU3WwhYEdaSPJfOTWVSbWthwNO3l2gZapXbf0pXajdA1ZjNpzS7QvNYVFdaF2JVUAUm4bsCs6cbj1ZYhiMhA61cFrtGTUp6+6JysrCPDxxSxRYPNbRepnd2JioIpjDFTNpAmIkgBqZZuaE1y2sOw55QR2zVSeShW2ZMV7xDJKQM1B5240KxqkOWUKaY8gc1EXuiuV3uQTDvUAJPerpRdwcroa3B+9inJJx1raUbaMm9ySMFznFWRZllyRXPN+9foNuxBLbhG7fhURTyzntV/ElYm7FLY5zQkuTirl8Nx2Zct8EZrRgKgA1moqV2yJDbmRQhyRWb5xDkik57IKUdHcnilz1qQygU+fWzL5SSN8nNTg5GTQ9rktWZNFgjJpXYYrnSuxMYATSfZmY1sm7GLdmONttFQSIc1M4tPUancqSnDmkJBGcjNOxt0uQtKd2KsW77hzV2vuN6Is8bO1JHweuaycbehKbZOMkVHNASMGsqs1pyjTtqVVjKyEdBVuEkcA0RqX0kjRsvQKQMmq95OFJraNO8W0c97yKMjGUGqj2+5uKiWmxtF2YjQmMcUqdetZSdjZNdCTYevarVvIUxwaFJ2siHqWmmAXrWdezsSQK3cmpaGKWupRLHvmonGRWbblsaNWKk0eM5rD1NsNgV2UNXoYVDpvB8mYCG/CpfE1kJ42bGc12VN9P6/M4Z73OHa38mUg8VJa24nnAHSm5aXOiO1zsNKtvIiBrQWQ1wylZnPKVyVXz1pwyetRJpbEX6k0MjIRjNa1vcYTnmiM+V2ZMrMlNxxUZlJapk3exkkSCcKOoFH2xU5Liqg+XQLXGvrsEQ5lBNU5PFMQb5WH51rOcltv8A15jjTbCPxYhkAz19629P1L7UoOawniGpbF+yaNuyl5zWnEhkIJrppS5kjJqxfUeTGSeMVyfiO/Mjsobj2raq+WFiYb3ORupTnk1RlmIzk1xROhMzrq9AB+as2a+Geua0SvoawRE10xGQ1UbuU4yWNXCNnodKqPoVludg5Y5rX0TVjFKoDY5702mtUbxfMmjuLHUw8YO7Pvmny3obq1Ks21p/X4mHJZkS3AY9acJAe9Z20AcJfU07zuetLbYTQ4ygjmojJjnNcteWg4jGk9TUsABYGvNmrtGj2NCFeK0bLAwM12UleRzTNJHUDrUyOD3rs5ddTFjZ0DqTWVMJFY5yMH1rog3Ej1KF9f8AlRnceBXI6vqfn7vmOKzxdRu0UZSehympSEseaypn461vh4+5c5G3cos5DGnxuSMGuzY1g9SVTz14pHkCnrWdtTeLs7l2xvvLIJNblrrWQBuob7/1+J6CipRuPuNQEik7h0rB1CYOx+apSSaBR7mepy/Wt7SH2Ad6Vb4bHLXb2L92+5M1FZqPMBNccF2Mac2b1vLtTNOl1Py1znGKxnTbbsdEHdjIdezJgtit+w1PzQOc1bpJKzf9feXONjTS43L1qtd3TICcmvOrRe5nGOpi3OrshOTWVd6wWPLZohSUkjSTUUUjfszdeKmiuQxAzW0qdkedzczL0D7hVlema5aisCZIhJHJFXLNgrg5rnbtqjRM6nS5QVHNdFp3LAjHNTT1mmhy1R0lryvFTgZOa9+MW0crMzWnQIcn61yu9S+c1lUguZeRHQnSUHip0IAzSve5DVipeXJxgcY71TWck8muN1H7W9x9CzaXJikBJrft78SRAEjgYzmvYw01y6scLtlS+lWQnbjFYl0mWzmuDG04z95nVHQWyISQH3rp7G7jCL0zXPhmoKzFPe5JdXxZAN2fWuX1+T905JJ4rtk3OOhrhE1URx0Cs0wLDkmus0WPcBweaeEi4ux6uOlpodTZ2wwMVfEWBmvoIrTU8eUuYekeBUq8CtDNmJr7sE/pXFzS73JPrUyepvT2Imb0pM89apW6lyQvXvSc55pabEjSMUqr71WuxOo8A0Ec0XACDScmmxCcmkPNIGfNRYk/eoIPrXXaz2AcGyOTTom2yAg0NOLHHTY9M8C3+4Rg9sCvVdKl3wqc9QK6IbakVNzXTgUkke6tktLmLMrU7ISKSRXC+JdHDZbb7cVEtVcqnKxyRQ28hHOKkLqwxnmvHrwd2z0aUuZEbRnBNMKFSM15t7SOyLuOa3YjODVd45I275pqTvqaqS2ZJGDjnqaeMg9aqbXLqVHct27e9Xo0LVzNvYJb3JPK25qOT69KnpcabuCMcdaeqFjnNSl1G9HcvWnGM1dJwvJqOTl1RLepC8wpn2jbnvVJa6CtfQPO3HrUudw61Uou4mMZRUbSkNwelW0ktQj5jo5CxyTSvFu9K5JNp6GjVitOm0Y61TZigzjmtqScndjWiGibJ5qRFL/jTlFXuW3oE8LKpNUmdhxWkJXZjzIjJJPNSRD3rSo7rQqJaicDirQvFCYrKMbRSYST2K7uWYntUM2etNJdRdSPnFImc5NE2rGkdiVJivQ4qxHeOFxms4SV9QlG4jySScc0JE3U1m3zMFaxJjFNySetJrQFuWbdwOufxqR5wT16VNpXBrUfFcZOM8VZU7uppS6MykTRoBzUgbFVGWuphLURzkc1Tup0iBJIrpqLZ2JhvZGJd3mW4NQ/bmXvWWvMd8Yq1mJHOZG61Zjmde9a7EyXctLcFsCp42LEVz1dGQkaFrFuGTU8kQIrjkubVEyKpstxyKcluUP0rZUno3/X4E+16EruyJWZcS735rudoRSRnT1YqqCM5qNeHzXHUjdJm6HSRlvmNQrGQ3TisKiu9DWOxKCGO2rIQKvFHLcUtCGTcBwahaFmGe1axd9GZtorPFk9aTywoOeatxWiRSfQoXSlycVh6hbfNnmtaF4ysRURt+EpNuU9q39RiEsB9xXZUvLVnFNe8cLrNoYpWwuBUmg6e8kgfBxmi7ULFcyUTr4YdqAYqQRVxS1epz6EiQ46VKIcDJqXK6uQyWNR1NTxtzxUJN6iJgpxzSSHykLZrojTkyGzm9Y8Qm1YhSc1gXfjGXkbjn61vQp+0bv/AF+JpZWMyXxNPK+QxH41Zs7yS5OWckmtpUYwvJb/ANeY+bqja0y0eaZSc4rutIgaKJQQRxXjV60pS1NYu61Oi0+QlgMGujtEwgOK9PCq8bs5auhDrF6IYSM1wOpX++Y5PelXl7yQU1oY93cAnOax9Rv1jQ4NRG2rNOU5u71MknBqj9sYtya2jB2NUy1FNI3brTbhWcZyaT0N4dyt5O48mrdjEyOCDROWh0097nSafeMgAJOK0PtLOMZrGU9NP6/EckiSGVs8mphMRzSclYz0bHeeepNKJcnrzUX6ksd5vvSGX1NYTjzLUViMylm61esyTjmuGcHz6GkvhNBJ1TgmpluvKO4HNdVNNao5miGTXikuzOOa19Pu2nUNnOa7YqVua5lJWLxLAVXnAKkmo1vdmUrW0OJ8V3rQA7Twc/0ribnVvmJJoVJzlqZTehk3t+Hzis6W5z3r0aNNqNjlktSHJY5zUsZx3re2hpReuo5pdoqGSTPekkbtjUmx3wauQXJC5yf8aJo7sO7oma+cjG41A++Ruppbbm8mootWdiXYMR9K3bSyKKDiuLETfQ8arO8mx8seRjk1JawgEZBrCEmrMI6GrERtxVC/OCcGtKkHe6N6d0zMMhSTOa39GvyuMn9aSff+vxNXO6Ons5w6Z3VLOglQ1zVkmRzWZzesWzKCVHNcrcu4lIJNZ4Zau5z1ajbsLCSx5NXYGwa2qHPbW5qW7cdauq3y159TXc27DgcDNTQSEEY61zyV7stHR6M7nHWu00aMtgHvRS1aZMtDq7O0BhU9Calnt/LjJr6GMOWKRzc9zkvEk7gMNx5/+tXOJMQ1cNdtTsmNaotRycZzT/tDdM8VlKTi9CVHW7IpF3DNQ+WQaxjTfNeREnqSKAOpqZbkopAbFekoR5boule5EbtmY5J/OoXdmrnm7wsdltRq7lbOTVy3vWjHJNcE4O6fYbSZP9v3DrWdqcpljNdkZc0bFU/dkjnkwLj6nFdboS8rXfg6fLJM6sVP3bnWWKAAA1fEII4r3Y6o8tuzDysdaRkI5osCdzn/ABAhZMc965X+zJHPFYT5r2OmFkg/sebvzUUmnzRDLDrVa7sptES27uflBOfSp00ud/4CQK1jK6uiXIZLZvBy4xUQHeq5XuyBQpo24NJpvYGxCDmkxTtoNBt9aTaRz2pLUhs+aB15FGDXW3rcet7BzmnL8vNJgdf4H1HZOqE45FezeHrjfEh3A5xj6VtTl3Kqp9DpYTkA5qbGa3TurnNIrXMW4cjrXP6vpwfIwKe5C3OD17Q2DsyqePSsKKwuBKRg8etebiNE0ehh5PW5e+wuqZKmq00RU/SvDnLdno07EsUqlPmxkVDMUJ7VtFKVhtNMrs+04FPVs9amsrbGsUWbYjPTmtCFjtrC1ndmjiK7Ek81GwPU1F9eUaVmOjXByT+FWYiuOetaJaET12JhKF5FBmZ1qHo9AjHuRruLZJpGfk5qLvmLkRPciPjNPgvdwxmtuay1Eo6XLAlBXg1E5PJPFEpe62SoiQ3BVsdqsm5CjpzXLF8z5jScSnLc/MSaqXM/HSulQ0CMbsrJIQc1chuMYzWaae5pKN0WJpwycGs6UZ9qU5KOiMlAYIyehqSNPzq4SurjSsTeTgZphJSmpOSuJ76ipITTWyzUXAVuBULEg1DSbNEhvm7T1q3aFX680mmr2RTXu3NFIhgHijy89BisUnds57jXgPU9KiKEGqinb3hxmhSSKciGTvRLV3RdyxHDsAINTQkkmslFpmUndXZaiyetTBa1ULM55PoJIOKwtWypPNaSkyqOkjFldvXgVAZWzxT0udqLNo4HJq6rB+hq9ErETTuW7WIdTVlvl5Briqc09SOpbguRGlPS7LvjtWCk07P+vxLcNDQtwrJ0qNh8/SvQjZxVzz3uyteN8prOEXmPzU1TekrIn8gD3qMxDdxwK54SeqZomyXaAvNV52Cg4qXHVFRepWSXY2SatRT7+9JRabTNWrxuOY7qkUKEIOK1hFbs5Z3RQnKiQ1E754rSUUXruQyRDaTWLqUZ5q4Qd0xTempHo14be4AJwAa69LgTQAhs8V21X7uhy1I63RzWvRLuzWh4a8nyTwM54rBe/Fpmck+U2tgzmlBArjcXsc7uPUgc1HPfJGOTVwjoCTZXh1MPLgVr233N5q1Ta2CasiO41JICeRWNqviIBCoPNbqXLot/68wjDS5yd1ctdSk5JJquPDV5ftmKI4PenTqezd2RLc19L+Gt7KwMwwDzwM10Fv4EWwAJyT71OIruVNz6Ec+trm/pmiJGykLnFdDa2XTivJoRcldm/PZmzY6eFwSK1GkEEVe3QXLE55O8jm9cvPNyA3SuQ1EjJPNcdSXNK7NFdbHOandmLODXMalqLOSua2pLSxpF3eplvMSMk062iaZq6bWVzSMTbtrTYgz1p0kIOfSuacrtnVTi27Ff7ON9WIowrVnK+x3qk0XoG281bjmJPNRqtGZyiWFmI6nrU0cpPNDbMWiTJPOeKPMAxzUxbsSuwjXHvTfPpTta6GloSRMGI5rSt3CJmuZx967FLYrTXrrOMVpW0hmi5PJroa925hPYjbT2kmDV0Gkp5Cc960ptpWMZaml5gNQXH3T6EVTMzhvF0G9SeTivPr60YFjVxurSZjUuYt0hjbmqxBc16VN6XOTdkqR/Lmgcd6Lu5rHuNZsmmAEmqRtFXZIiDuKnROOKiUjsT5VcsQ2xerkVqARkVzVZmNapc0bRFXtV9ZOMCublUnqcij3Ezk05W20lA0iieOf5ajuQHFXVk7XN0rGZNEd2cU+2meFvasebqRtudJpepEgDdW7b3IkXrmonLm3/AK/Eq427s1uEPHNcnrWisjl1GR3rmi+WasjCrDsZBhaMgEVPD8xFdMndXOeLNSBto5q2kmRXBOLbN0TRsXq/YWbSyLkHBrnm7JpFJdzrtI04xgHHIrrtJhyVHTAAJrWjHlaiTNaHTWzbEAOeBSXMu5cV7XN7mpgkcj4gjMiE4zjrj8K5h12NzXDiJXaXULDhcBRjNOWf3rnlK0riaJRPkUjSKR1qoy5tzKSsQGUbsBqmRSVzXVSbszaluPSIZyac0IPOKUqZvfURo+KryZXvXNOm7sqLuRCUhsk0y7ulERORWtDbUt3bVjGtG8y8BHPPFdno/wAqKeM16mGtKVi8R0On047VPPWtOJgQBXtRicLJRg9qVo8qcVMkBg6zD8wGPrVCKzywG0c1zVNZam620Naz0kS4Owc8ZrRHg0XI+5z3wa6KcL7mbqNFi1+GyBt7Jgdf881rw+ArdIsfLkj0/wDr11RptIydRM5rxT4L2QttTp+tec3mnSWsrKy528VFRcqNqck0VyhFGOM4rNFNIQrSbcUmAm2kZc9aVrMW58zgYpC2OldF9bFiE1G7nFNbjsa/hq88m7TtzXtfhG98yFRuHAFbRbTuE22rHb2bhl69quLz14reLbRzu42RQRwao3NuG7U+mpN2YmoaUsmTtrEuNISNz8tefjaPNC61Z00ZtPUo3duFUg8CsO9hwx4r5+VNrRr+vuPQpsovESeBTHjYCq5rI6uYhdcHrSYbOSeKHOV9i4vuWYZsEYq/bvvPWlKWupZbAXv1FRy4xWbsve6iTdyJBluDViMlRVN8wNimT1IqSOQORzmpcfeGtCZsBc1n3VwVz1pRhrdkrVmVLdO7HkiprRpM5yfzrbmWqZ1aLQ2bJGYAsasXMaInWpcW0c0pPmsimpG7FLIwx1rCmuXQud2UpZeuOtV3YscHmtak7KxUNCWG2aQ8U/yXjbGayi76oty1sx/lvt5zULIScGoqNjUkSxx5HFPWEg5NawXKuUy5rMkxxioZQAapKzJbuxqGnYGeamKumLZjJHXGBzULA4zVNW1NehEULdqsWuUbmpU7opPSxpRSjaKsRNvIzVQ1RhUZMxAHUVVd0JPOTWrpxkjni9SIkZzninrJt6VyKPvbnRcmjk3454q0uMU2rakVCaLIOTVjOK0UrnPIjl3bc1haqCSTScH0KpPUxZicmq/OfahPS52IfCzZxyK1rCFmIzUTfRFt6GgUMYxR8w61l7TlTMExynPGas20RDZzmuTn95I2b900UlKrijdnrXfDazOFqzI5gHFUvL2sfeom2y4OyJ0TimSJ83SoloNsGTC5rNvJSucd6u1yo7mdLJJ1zWnpg3rzU1Wjd/DYvNCR2qMIS3NYqUjJitZo4yRk1VktMMe1UuZy/r/Ijm7jDbgjmsnUrfrXfDXcxkzEnUxPuHr1rY0jVVI2Ma6Y6q39foT8SsiXWLNrmIvGKyNKuZ7O4COCBUKny3SZEXeLR1sNz5sQYVBJdtE/Wo9n0ZhbWzBr5inB5rJvbmR271TjyuwouzJdMBMwLHvXVrcqLbrzilKSdyaivqc3q97tY/NxWA3m3ku1ATk1zwk9Wwb0Op8NeEPMZZJVzn1r0DTPD1tbICVXp6Vkqybuc9Rtl6Q29ouFAFZkwN3JwvFY1puUeRdQpq2rNGy07aoJFaVvbBWBIrroUeVK45S1NCJlVc8CqGp6gBGQprurzUYXFGLbOUvp2YnmsW9cNnmvHjU5nZnVY53VIjIDg1y1/bEEmuujNX0Mm7My3PzVr6VEqqGPeu6afLY6KbuannrjA4pjzACuPl11PVwtO7RVaX5s06CQs4A70+VnsOmrGiqMgB61NEy45NQ1zI82tHsS+YMjmpkm5pM5miysyhKqzXSoetJ6qxnFMiFxuPBzUqSZxmueTt7pXQvWkTSHIHFW5HaFcEGs0pPcxlJXsQR4lk96v24MRGa6E+XQzZq2zDqetXluNo4OKu9ncwa1JornJAzmpJZRirm9mT1Oc1+3EsbEjOa8+1hPJkYYpaNpXM5q6OYvGBY1VVcHrXo09InNJdB5fC4xURq4lJCiMnmnrEc03LU3hoTLDnkVYggOcn8qxkyZ1ehehTaOlWI4ZJThUJrmkr3ZlHXQ0LXTLhxkIatjSrlesbflWHtLSsDQ8abPjlcVG+mTrkkE1n7fXQL2IHWSI4Pal3lhzVSnzXOmErjGiL81E8RXmoTV7FyVyS3uGjbg1vabfk454q0nb+v8zLbQ3rWYSLSXtks6E4rnqRVtN/69SZXOU1XSHSQsorMVGjbBpwvy6nO0k7FyB896uRfMawm7O5UUzZ0nTWnYZ6HpXZaRomwB9oxjpWXJdtlX6G/bWewDFb2k2px0/wDr1rh4OU9SJPQ2Fi2r0qtcZBI/ya9GzitTOJh6pD5oJ5ya5XULcrnjpXPJJprqWlcxJ3eJ+9T28pZRmsJU77hJKxbVgByarXMrD7uan2dtU9Tn2epRWaUTde9a9tOWjBbrXclamrlw8idZh1p32kZ60ua6SN2hWmUjrVO4nAFTOK3ZcUUmmPrVO9lLrjNckH71rnRFakelxn7QCT1NdrpY+RT+H1r2cAt2jPEPU3LaXArRhuDx617UXpY5fIuQNu61dVMpxVKN3cze5mX2lvcT5HPbipLbSTEQSMHFcjptyNHLSxvaXYDIyOuCf0rqbC0RYw2B0r0sPBWOaUtdC0VUUg5BrosSUNSsUuImDDPSvMPGOhLCTIFxis69PniaUpNOxwEgwxyMH+dRMM1xRjY6mxNuKCKegDR+lBPPWi1wvqfMe4k0xic1rpc0ADueajk3Z6cVV0yW7OxLp85iuFOOhr13wNqO6NMsCcVrGVlYrpc9P0qbcq8jpWsvI4rduzsYDgp9KjlQn6U92QynLBk8jrWZe2IbPygiocVLRoadjn9RsioPArm9Rj8o/d615tfCK94/1+B1062tjJa5AYg1OqCWPIxmvOq0VHc9OErq5UuYyp7VEKwvrsaoeinNXIPlxzSjbVFXLTFgmQeaqSzsOpzUp62ZrTsx8UuACamNwQK0VtkElcieYmliuShqLJTuNLQtfaWZMGqE5ZyRmt5WtoZxSuQi1zzV+0t1yNwrn0Rs5aXRpIdnI4xVS9uya3UlFamC1ZUiuiWzU0soZflPSubkV9DeRSdtxPrToYwWBNZ1X0HFPoadqoVc7aWSME5p3S0Ie43io2hBOe1KNm9hO6FUiPnFHm7m7VpK1rkNX1HOdozVVmyaiPwghVUEgmkk+tVT92NhNO5ASckmlWQYOcCr3NBAw7U6M88CplTikOHctxsdtSrcFamL00M5q7BrknqarNMWbAzzWnNbW5CjqOXdjJNOD5ON1YOoty9y1FIAKspcACpnLTcUok6XGQBUwm9TV05JoxlEUyBl61QvYhKMmtqkk4mSVpGTLZA54qA6c3pXO+lv6/A6lMmttPAYbq2bW2RFHAocWEpsdMFXmqjSbiR2rKVmxwHRcGr0Em0c1ko2ldm0tUStcAjik84seK61JHJKOpJHnHzVHKfm6VnJJq4R3FUkjpTJGxyalST1KIZLg/dFV5bcSdTU+01si0rFWW3C9BU+nKUcZPFDdr3LjI1yBtzUJAByaKcbvUyk9RskqotZ9xd4JOa6KUTN+ZVN8cncap3dz5vcGt6cEE4mNqB9KowztBLuziuqPLa6IidXot+l0gRznitVtIhuCCqjNS9U3/X6nLWbiyZNPMS4AqGTTTI2StYrf+v8jFz0uQTWxi4xVGa0LH7uKVWSuJPS4+GAoeFq4Wl2Y5rkdXVml7rUoS6PLdyDPTNbmieF44sORgmuSrX0sjCbbeh11lbrboAoxVz7TsTPTFY4d80nIhlJFkv7ggE471uWWlCFASM124Wk5XmypytoWhGEHSmPLt612qXIxIhmu3AO3NZN9M7E7uKxrSvozSNkzD1Kfy0Y5rl7jVWecr6GuKLSlY7KULq7IZ2Mi5NYt/DvB4rek7MwqRsYU9owk4FadhEVQZr0pVPduXh0TvlOTVWSQljzWVJX1PewatqJk45qW2l8uQPjIB6Vsl3PT3RoSXqGL5epqOOXA+9WEotPU5aq90lWf5qsJOB1NZ9NTzmrskF1xxVW5dm5rOS21M2rC2gLuBzWnDb/ADDNY1Iu90ZymdLo9qhj5HNWr2wWQdBUVbpaHnzm+a6KkWm7G4GKsGzYDIrPn0uv6/AaqX3I2eWI9DxT47pyMGtYYhP3TTQsRXBHQ1O1yQOTVuq2r/1+ZBTvnMyEda4nxHp7EM2DzWsXFyXczltY4y7tHDYAOaW20WeXB2nmu9VVGOpy2vqy0PDV03RDTk8I3ROWXFRHFwNYxfUH8LXaDOwmo/7GuYz80TflTWIjITuiaHTpcgbDmr9vocznO2sKlflIUXua1h4aldgSMg11OmeFF2jKjIrz62JTRty2Vzo7DwxHxlRirr+F4gOExU0k6q1MnJXIG8Kpv4UUS+EFZfuc1UKbeiDmsc1r/hYwKSqdOuPSuKuW8iQqwx7VsqcV7r3KT6ixSow60SDd0qXGzuzaL7kAXa2au2s3l45rRcy1FNK9zWsdTGRzW9bXAkX60px0/r/MjUS7s1nTOOa53UNJ2Ettrmejf9foYyTKAgMbYxitXStMe6kUbeM0m02mEW0j0Hw/oDBV+Qdq6yDTfLAXbXf7G0dUQn1J47Fs8rXQaXYhUBx/niqwtBpttGc5dmXZrfCZGKz7i3JGR1FaVYDgzJu4iRzWDf2vzEkda5E027o0bMi50xWOSoqo1l5Z6VCV/wCv+AQ3cYVKnmm+UGGTTjTVznkrsieFVOQBmkWXZxWlT3Ua4dah9rI7037XznNcvPp/X+Z2W1Gm9aopJy/elOpJqw1oNAZuahmQj71Z+zd0+pvSd2TadETIMd66/SUPlKSele/l60aMcQ/eNaJStXYDgCvV0scvU0rMqcc9a1rMByFPOe/5VqktjKWhu2OmRuu5gOfepp9JQH5QCMf571caCsRzNjLW28t89h+takMoUAZAx61pT91aierFaYZzkVGboADB4NXzXYWK11qUaR8tn9P61wfjLV4HibLDPcE96XtFY0hHW55hOd0jEdM8VAxwa473Z1ITcT3pGbvSAAe5oJ9KG9NAZ8wmNutN2HNdEGr2KuOwcCmuhPJNS2tQlsRx5V+K7zwHqRWVFLY5qlJ7scGrWPY9CujLCpznGOK6S3ckDOK6U00jGVkWcE80hUmr6aEdCGSPv3qtLDuFLzJbMu+sN2TiuW1jSDtY46jilUSkth+ZxWpadJDKSaktGdF5B/GvIxisj1sLPmjysmeMS9RVd4CrcCvJk+p6FiaC2JGSOaew2NnFVFxb0JktR/nZXmqkq5bNDjZmkHYfGhwKn2cc1S3Kb1EMWaZ5R3YNQnqUSHIGM1XZ23VftE9AsiaPnrVmNsEVGl9Q6E7TjFZ15KCSdwyKc9TOOjMx7wxtkGpEvzgZNSr7nS0mhRcbm571ftPmIonFJCTsacJAGBSOSa5JaPUza1Is881OqKVySK3pq2qFNsrTgZwDS28POSacktiOhNcRrs+U1muCrHmojo+UcOzEaUikWU55q3ZGnLoEnzdKjERPOaXPZ2RKWg+NOeaccDpT5tGxrcmjbPepShI4qY7XFN6kZDdCaQDYeTTnyuNheQNISMCkVSrZNYRSvZj8kS+ZjuTUiOW5JpTUkrjWxbiLYGCasxhuM1MZu12YyJecVFKpatIyvozJ2RUePDc1JHErckVT02BrQiuP3XNRpeOo609kaQV1qQz3bNyxNJFcL1J5oj1uaJOxOsu48VdRiyCuStJ3saNaXELFe9TwuO9KnPqYuOly7CgYUPCM11yS5bo503cBCADVa4jwx61hUV43RpCWpXEQLmo58qMLWUFdpGrYzyty5PWkjiZWGM1q9SObUvxMQvzGobiQLnLVCck9AbMy7vM8ZNUbi5+Tg13wVldsTWpRaVietJuxzmh1GU43RUu08w81Te3+bArWM9COXW5bsZmtRmup0XWlkwrt+tbwkjnrx5lc6KGWOVQcg5qQonpVSpq9zzmmtCtPbIxzVWa2QVw1IWvcpNkGxQ3QVPbRozcjNclo2aNLaGpBZpjgCtO3j2KPQV5E1HVXMSwGxUF3MduK1oVOSDYktTX8N2w2CQ/X61vPhRXu4eNqKb6kS1kVvJluX2opP8qsx6Ptw05yf7vataOHdaWuwSlylLU/JhUgbVrmL+5DZxWWYOnH3Y7mtKLerMO/zKpFc/Jp483dXkU3rc7IyaWgksRReKoTwbs8VupK9ybXKRsQXyRUnliMdK64vmNaMLFC8uMPjNVgCfmz1rupJRR7mH0iTIvHJqaGJSOpqmr7HTzOwrxbT7UICD1Nc1SbYpu8SZFYk4pXLjpWC0d2efJXFjmYHDVLu3Ac0VGtGZdC1ZIN3GK3LG280g5GKltNWOOo7Jm/bgW6UNeIT8xrlqz3icVmxYZ1dvvCtCMB16VlHtYzlGwNZJJ2FRNpwU8CsVFKV/6/IfMyMWhU8USwMFreMnaw1IrCFi2CDim3ejpdwkFAc1UZN7jb1Vjm7vweqy7ypxmtDSvDcYILIMCs6teV7EtK9zet/D8JwdlWl8Nwn+AY+lYuTkrWFdXJE8MQuMbMg02fwRG6E+X+Yq4SbfvbE310M6TwXHFJ9wZq7Z+FI1PKYqfaRnNouV9DUt/DCqQQoFadvoRReFJ/Cm6cpPuJyXUm+ymAH2pRN2NdtK0UY9S1bhWIPFaMMUbKMivQo8rM5XWxkeI9KimhJA6ivDfHGmtZ3bOo+U1FanaaaNKctGczb3bK2N1XFvM9TScE2UnqTLIDzSvPtTNZdbG7bcbkMOoPFJ14rqdG1YMo3NVtJ7/1+JK11R0NvOso5NPlshcISBmuWqk9v6/MmSfUz18OvJMCUOM9QK7Xwr4R8wj5O9aYKh7SXvI56unU9C0zQFto9wXkVNND5edxr2qkIpWZim2UwQHyD0rX06cY7Y61y06iWgSRYnlPbpVOSbgis6k9WbQWmpm3eMZ71iXgya45S1uh26FCaPrVGdRmiCTViZLS5QuE71BvwK6KSTdmc89iCaTvVWWbaM5rnxCs7G+HXUqtdYPLU03Ixya51bqdthrXgXkmnQ38WeWFVFqb/r/MTTuXoruEjhl9+abcSI4yrCutUY8nMXTve5Y0pAzg9RXYaXETGMd678CrJnPXd5GokJAzipY1II4r0JamCepdt85GDWtZTBSpJ71qnpqRM37XU0UD5h17GrY1FWGdwJ75raFRJGfKyN71OzAfWon1BeCH/HvVe0T3DlZG2sJGPv8A5n/69Zd/4njhQ/vM4464rNzte5caZyWr+N2cMscjH6GuS1HUZ79y0jEjsDWTkr6m6VkUGzmoJGw1LroXHXQYX70A55Jo23HawoPeg5zTbXQR8zMxK+lRZI961glcryAMetPwdvNVNLVoHsR5Gc5rY8NX32a8X6007JIIx1PaPCWoCWJMN9eevSu7sJdyA89MV0paamcl2LyZNPINWjO9iN045qFkyKlifcryxZ61mX9isqnjNVfsK5yWsaQAW4/SsB7Lys5GK4sVSUle2p2YWpysYBg8U9bcOckV85iYON7HtxlpckEGwE1XuBis6abauKZAgLHFPa34zzW0txw2GhSgp+7nkVSb3Rpa7uh4z2FMcHuKxcXuUBJIqIrzzRbUCWJT+FPclRW3IpWZmyrcXoTjPNZdzfliQDW3JrYqKKocyd+lSA4weTWkafQbl0JopPmBrQtrj5uO1E4WepS1NO1mDke9W5CqJnFeZWS5hVE7orq2X6Usz7V54ohJRdrg11Kwcu3ercKHHWidTsQ9Nh0pGDzVCZRuwOKVO922Wroj8okZphRgelXy6vmHcdsfqRTgCBWPNraw9LEchKgmoDO2eSa3hbl1JRLDcZar8EoOSTTiE1oEkgPTFVWLM1YtXdkKKHZCjknNNExJpwjqVYmRsjJp6ZJ61VRXQovoX7XJxxmr8aEcmolFNWMJsVye1MYHHNFOK2M5MpXT7DxUa3bDoOtXyotK6GSbpASTUGSKhvmRcRkhDDnrUcUeZKzTsbR0NS1thjmrqwYTjpXNK7mKTIH+U8mkSQg8VcYpahui5BckCrMbM5yTXcpJx5TlmkiyEwvPeqsybiSawqR0sTF2K2NpNV7hwpyayp0za4xp1C0+3ZW5JreSUVoZi3M21TisW9vnXJJNaUoJO5cZLqZjXjSPjJJpDuY5JrSS6F2IbibsOKiSYs1NRurFxVkO2lzzSMioOTVqGlkZz7EEsigcGoFv5LZtykjFdFJJLXcwa7mpp/jKS3wGJ/Otyz8ZLNgZyTWkoJbP+vvOOdPW6VzSTWlmHWmyXwYda4qrUlZGXIQC53NirtpISwrzcS1GN0UbVk5brWpGuRXhSd5XuYy7jiMVBKm40Ko1HUlbm3o0nlxAVuW6x3ABfOK+owkozpxgQ7p3NCG4trOM5ZQR781jap4jWSYxxYCV2YrFwo0OSG4UqbcuaRh6ne+YpZmHsKwLibJJrwZ3bvc7YlR2BBz1qlcIM5xWXJpcpXKU4B4xVKaI59K6IRuUlqVZPlNU7mTAJJrtjCysjrpow7mffMRU9uvyZr0YwSgenDREwyasRkqvNKSsrGoO240i5zmseQvpYtxEY6USsvXvWFWKtoYezdysM78k09ZcHJNTFK5y1YWZYgnO4c4FdHpc+wLzmk01qjirLQ2TOZEAFUruCbGVzXDVj710cSaT1DTGlSXDgnNdPbsDGMmr9m3G5nU12FNwqnrTZLxVrJRV7EqLFS4R+c5NTYVxmqlZEpMja3B5qGSURHFOKihx1diNik54GantYQprmqxvO9imjWtVFaMUakAGtadGLRlK9zRsrKNiOO/WtmPTopExjrXoU6EFHYxk2VLvQ4yRtU++Kjt9KVeCM4Nc1TDqMtVuX7R2Li6cqjoc1KlqFBGOvWtadNwZDldkV1p6SJwtYs+nFJCe2a0qU0lcqLdia2tz6VejgkAqUpRFzdypqUMjIQ2eRxXlfj7Smkjc7DkU6k3L4jSCPJLkm3nIII5p8dxnBrqUeaKaJasy1Hdep4qR5N44PFYNWO2ktCtIpBzV3TL9oHG7OKrm00M1LVo7DSdTEgHJP411uksswFc1tdf6/Ema01Ok0/TY3wT9a7bw1bxIFXAHofyr08Byp3OKsnY6sW6NFhcc+9Yd9YSbznpXXioSt7pFN9zNa1aNuQeat2m4GvLhFp+8W9TTitDNGCQear3FgY+ta1aTaui4yZk3cWBVJdN+0da440XKpysfNrchvtAmjA2jOfbmsK906aInKn8q2+rzT2C6aMm5idM7qzp5NuadNOLdzOpHoZ9xdYHWqMt2TzXFXmlPc6MPGyKktx3JqrcamIx96uRPm2Ol7mVf+IVjHDVkt4nnZsgkD616tGhpeW39eZhOotn/AF+BbtvE9wRy55/CtGDxK+35mJ+prdUFG7v/AF94Kp0Oz8IXr3rIeTnH9K9Q0yyxCh56V34aCUdDnqTvI0ltsDpQISD0rqlDTUjTcnjiI5PWqupastjFuLYA96mTSi2VHVlKHxevIMmPXnP9avQ+LlbGXPHfP/16xp4hM0cGWP8AhLASPmH4n/69VpfEjv0fP4//AF619or6snksZl/4inKlEOTWHe313Ny7tj60+a9y42SM92Y8En8TSMp6miOug9LEEg7mq0mAc9aeqYK5Hkmjd2NLqXfUehJoJptO5PU+Zzlj0pmOelb2V9C79BcfSms5PFFidOpG2RyafZ3LQ3CsOOa0UU2UrHq/gXV/NiXL9q9W0a4DoB+PPWtoSVtCJqzsbsHK5zUu3NaRVzFiMmOtQyJQ9wTRBJHx71Wmj3ChK+pLMu9sPMByK5zU9HPJC/Tiomr3RcHYxv7MZSdwwaU2rJzXzOYQlGX9f5Hs4apdIk8vdHwBWddQsTnFcluVXW50ddSDyynJFP3HbjHNdEY3d2NbEDsd1SRDd1xmprSVrI3irItRoMc9KguHXpwKm2gt3oRqufpSmPuBTa05kLmEUbeWP4U2aT5citYPQHfcyLwlicVRaNiCWzWynrqPoNAKnpUofjNdNNq+pLWo5JPerltIQc8VlXas2zWCNO1udoq+knnCvKcdWypx0uSJFt5NMmTdwO9YKPvEt6FcRGNqmNwqLj0rRw1RG5VnvBk4PWo0fcQetdDWmhbTLK4x0FJHEXfkcVlOW1yUWXgATJHFUZPlzVqMfiQou5Xck9aiMWRk1PLd6GltB0UW05xVjJUYrSWjViegnmkd+aY8+B1oSXUlbjEZ3GaXdtNZc6vZGrRMjFhnFWIAS3Iqas7WVyeU07JDmtEr8vWojPXmRzVFqQs+DUUkvHtW8EtWZuJn3cxz0qvHId2aGk3ozogkohLdFQRVKS6IPWtIRSWpUYphHNvPWrEUm1hzWM0rmvKbGnvvAz096uyzqBgYrmtGUmY1FqZ8snmSYxVpLYeXnpT5Vsx3skhsIUORmr8BAOa0pNWuzKqtSaW4wKg3kgmqfvMzsktStI/XJrOupSWIHNVyJMN2VCzk8k1YhnKpyaa99lsdLcKyVz+qT/MTngVsqdnqZ/aM21nLSk9qsT3RHApqN2daiimzsTnOakgJ3V06NaEt6FsMAMk1BOCwJrNxtdmFym0bMearXPyA04voZzdyiqNLKAK6HTdOKIGIpYiq4QsmcdV66GzAhReSasxEnAJrjUnKN2RzXLEaAcmrdqx3iuLEv7KGblnJswTWlDdAivPlTjFXMmiXzN/ekx81YKMdV3EaVk21OtXEvZIR8rEV6NCbhBJPYVlcpXlxcSqSXNY8kzrITk5rKdVzfdm8XYrz3LEckmqc9wKpK3UvqRM2RUTo7DpWidyiBrRmO4iqd0m0HiumlDS8v6/AuLuzGvJSucmsq8uPkPNdtNXdkdtLoZSZkl49a1bZAFwa9CasrHcyYgA8ClBxzWTTehqn0F6mkwc9aztY0i7Mkib3pZG561hKDcrlvcibI69KTOTz0qLHPUhcmhlwwrZsJ8DvSkrx0POrU7Kxs2t5gjnIrXtHjuBzjmsPZ30Z5dWnbUurp6gZUc1FK8kB47ULmpqyMU09GUZb193NSRM02Oea5Kid7ottJGhaWjlq1o9OcqD1rGM5PcxlIjurd4BzmsW5ZnfAq+ZppM0hqrlrTrRmBLHnsK0hZMq5xUzknFsL3LdnE3pVhpDEPeqoybjdMzZZstRYHk1t2uptlQBkV106zdkTKn1NOKYSgE96mEAzkV6UoOSTOdg3vTCKh26hYeqZHNUb+3XritOVMa3KcAxJitOJhjr7VKjaIdble98tkOMfnXn/AI0t0aJ/1rkmkr2Nabe54b4h08rcOV9c1iEvC2CDXXQl7qRo9UKl0xPertvOWHNXUhZXNYvlRZABHWgRsDxXKzPmV7m9onm7gBmu+0EugXIzxWU5XsNu6Ox02YnaOa6vSJHQjB9K2w85JmFTax1+m3AeMhuuB/Sr0ttFJGCyZz3r24y546s5HpoZd5pIdvlAAqGLTWQ84NcNak+a5o2jYsbbagznNRapCMevP+FdEofu9dyIy1Odu7Rj8xHWobf9y4yOOtcMFyS5jS9zS8+KSP5iRkc1iaw0OM4H0616VOrTa1EotM5TULcTZAHNczqNi6sW7Yrkxa0vE1jfqc9eROGPHArPlSY9FJr56fPKpc6aa5Y3Zn3XnqDwawb+eY564rpo0eV6/qZ1Kpi3LMc5NVwCDya9VO6scXNfcsQyHitC0RpZFHXJpq5opaHsXw50zAjOORg/yr1ywtgsanHQCvTor3NCFLUtfZsDpzQLcZ5GK2cbj5tR8sYjiJ6V5940viJFhRjjOTWM42i7s1pavU5tJ5CeCa0LSaQZJY/nXmS5dkdD0Vi0L1h1apo7veOScfWmppf1/wAEdrq5L5iGkZkYEV0KV3qZOL3Kkluu7IFQPEc4rpUxWbIZYyByKoznbStdlLciyRzTS2TSd00U11HhsCgNim3poNrW582kdeKYFyea1SbZK7gyetROMdK1W2o27kTk45NRhgpzitI7XFc7HwTqhjlVC2BmvbPDl/vReckCteoVLnXWc28D3rQjAYdRWidlZGOw54+KheI0dBMryqfwqExintsLbYglh3GqN1Y7wRS66juZF1pm0k7ck1nzWRHUA15mOwsZrmS/r7jsw1ezsyu9qQOFrPurcKTurwp0XB6/1+B6KqXM+YAfhTOCK1jruap6EEgGaRXCmolC7NoPSxKLnjG7pVa4mG7OaVnsXsyW2lz24qz5i7auyehnNPcqXEmDxUPL0balrYZJbY5I61ELbcaybv7yBa6kU9oqjJFZ9xGQeCa6cPJ35mWNiVyRmtC3ibAJzWmIV1eI0yyr7G5rQtrnIAziuKUWkrmrV0XFlO2mS3AUfSsUrswaKkl2WNV5p2xwK1iuaW+wJWK8bM781fhhJUHPNDdtUOUrFuOLnFWoYdvJrKcefUxlILhsrWfKmTzmqXuoIjPL3elL5HFCfvFtjShB5pG6VfXUaZWllIYgVGuWPPNEldXKii1GAq+9Kqhn65rmS3Zo0XIIVK5xVhIVBrNwco3ZDbuXrfCCpjOD0qb2Ri1djOXyTUFx8o61rSqWjqRJalGWBpfmJqrOREOTXTSkm7lX6FCe4OSarBjK/FbTlZGkNHcvwR7V9TU6KSelctRaaGqZpWm5V61O7kjB4rmguVmU7DEB3ZqeS5KR4B61o2rNkwSbIYyxbNXoZdoHPNZQlr5Dqq+iHPMD1NQvcHGK9Cly21Odoqy3JDciogVdstUyV72YWdiG5KLyCKpyznsaulG2pdtNSCa5fHWsu8DSZ5NXzq+okupWhR0PSpDGWfmhb3Ru2hxg45NKv7sVcJdEZ/ELvzyTSPLgdKJ6vyM2tSGR1AJyKzb6QNwOacFtcxluWNDsfNk3t26V00UYQAYFcmLk3OxxTfvDpZVjHWoYNQUybc04Jt2COppxSZUHNW7ViWFcOM+LQm5rRHagOaswTn1rza1R3JtfUvQTZxk1YzXLB33E9C5Zy84qxIciuqM37MT0ZXuJMLWRMDk5pUaltGaqSaM+5uFU4NQffbPrXZa6ujRbF6ztEkOWNTXcENtHuJFehQw0XT5iHJt6FD7VG6noBWJqtyisQDW0qVlojaKaZzt7JuzjpWLfTHpV0b3O+kMsU3Hca04yFrtm7u51XuOyepNIHx15Jpb7G0WKzc9aUAnmomlY3iTRrTivPNcs3royrjWA5FQuOfas7EMI2w1a1i5bGelCucldKxpwcdzWhaXjQMKbd20zy5xvozesdbR8BmGT+FaIWG7GQQSajlUotHnVIODKt1oYYZFV4bGS2k5ziuWrFpaIXPdWNjT3UkA10dnGHTpwO9YUo+67mc9yHU7NWTI61y81mVuPqa58QuVo0hL3Tc0vT/kB9a2V0wsOnB6Vs6LcDPnEl01olzjFZ8sDFitZQi4rkGn1NDT9GMgBPXrWxBpXlr9K64UnGzJc7otRARHkgVZSZWH3h+devTmuWxjJdRjnnNIsnrSktdQRJ5wAqhqF8qDGRiqUPMaWpiPqkccg+cH8aVteWNc7unrXDPEOMrGvIZt34oUgjeB+Ncn4j1sXEbAMDn3rkddzd+5cI2OB1C3E8hJGazpNGEmTtAraFVxX9f5DuUZ9E2HIFQratGcba641XJBfuW7a0kkP3Titmw0R5iCy/pWNR9iJHVaLoCowyP0rr9P0xY1HGTURSlYXkbthZnIH411OlRKADj0Fd1GDSuZzdzftX8vBB4rQgnd8c/WupS6GVjVt7cOoLDt+tJLaKSQB1967PZpq5m5OxYsbElV3AHH61BqVmSoByMDH0rWpT9wSRjT267c449z0rIurcLJkYryq0Uo2NaW+ozymIxzVC70tpwO349aqguprcZa+Gg/JH4mi68GpJHnbXpxoqcTHm1MS8+H4PzlQAf8APrWXceBo4j8yAfhXO8HBO9i/aMz7zwZCUI2CvP8Axj4WFpGzxoMjPtU1MNFx0Hc85uo/LkKntVYjHNcHLYxbFiJLjmuk8NWv2i5Q4zgitoJtaAe8eA9P8uNGAxjofyr0Szi+VVzkY616VJNQsCbLixbjzTktct9K3VmK5S1eQQW+SQCB0ryXXrj7ZqUjjOM4FcONnyQudVEqwxZq4qFUxXlQbndm0vMhdip60+Gcgjmi9mUndFuOXd3pXZlGRXXzW1D1FjdpCBV2308zZxVxk5ESsidtCd0zsJrPn8MyM/ANdOyRFxy+EZGTkDPvWNq2kPp7AFcZ70ODWrY1K5nBGPrS7H9DTvYpPufOC55oGR1rpejBg7Duagc/lVLcCItk1CVIbJ6U4qz1E0aGjXZt7lSPWvafBuq+ZGgLckDmtVq7F7xsejadcbkz3NbNvIDXQr7HPrsyyrEjrSsueKbSFykEkG4moZIcGk/IWpA8JBzULQ+3NS7hfQrT2u4YI69Kz7nTzk7RTcU1YFK2xQnssDnrWFqcDLk4rxsRhdf6/wAjvw9a71MO4Xk9aiVc5Oa82alB6o9SDuiGZSOSarTSY+tEd7mlnoQ/aWA96jLSSnvinOpFam0VfVmhZwNgdamlQqpzmseeXNcU0VXJJwamjTA3VVS8kPZA+GHPNRbMVKVtEQpFe5PGDzVAwliTit4ySTNYJvcngt+ckVdRPl4qXPmTHaz1IJUYv0NWrKNt3INJPmjY1bsrGtHD8oxVe5gJBJrhk2pXRzuWpRKbSc0gtJZzhR1NbU77sL9zRtNAlC5bvViSxMAxT5Zf1/wxk6ibsJEpXrUpkFVGBnLcjlO4VXZMjmpknccdBojJOKcY9ozmjkd7luRVuJQmTmqjXeeAat7mkFchZyWzUkZ55FY68pvFE4HGcUqhl5FZyk7WHfuW7V3Ygdq1LeIEZas6tRK0eplJW2JiuOlIiknJrFu+hmThPlqtLES3NaN2ViU9RDb/ACEmsjUbds8VrTmo6CVrme1tn71IluIzmt3J3uaxdyxCea0LdABk4pSjf4ipFrzQEzVV7lt3esHyxYlG+5LBIxOfSpC+7rUKMpbBJcuw9WwOKfHJgnNNpR0E9Qd8nrTGlA71pGo2lcyaKdxMQfaqzXDZ4Nap32EkRlmbrk1DMcVrFsb3I1XceeaV7cEdhUSu73HYhe2CjOarsoUmim2ib2GK+5sYp0mAM5FbRuyW7FSe4MfQ1Wa6LHrW8YtLUm6SGSSZHJqoT5kuBzTTuzGctDpdLVYIh0zirjzgDOa4ajUpnFYzr284PNZovmilzmuikr6slM3dM1USABjXQWbhiCDXFmMPtL+vxKZsRtlKVCVbrXi1lopEF62lxjJq9G4euR7CepetYyCDV7ZlaIyb91kMryW5kJqleWWxcmnh6Uqk7/1+QlKxxupPL9rODxmrFuTtBNezF2jZHYmrWL0M7RjINQX108i4JrtjUajyija5jXV4bZSc5Nc7fao8kh3GuiMuZJM2hpuVZbjdGaxbpi79TWlNWkztpaalmzVlq2hOcmuiTd0dC8iUkkZzSgGiL0Noj0HOT1qRWAqJRuaxbZIrd80eYMHJrlqRNBpfHOagkbJzmkttAQkJzIBW9psW9eKuCsrnHitEaXlMgHFIxOM1LjZ3PNbTIJbqSIghiPxrQ0nxDJCwDscZ9aydo3fUznDmidhpWtRXIUMwzWtNaxXEe9Tk1MpKVO6PLknGVinawMs4AHeussLBxCDg1xYeEpXaFOdtGLdQHlSDk1ny6KXy4XtRUw7qN3QlKxJZp5BCscDNblnco3Ga0pVIqPJIiXkWplR09zWRNbLvyBx61nWaTCF9maOmssZHqK2VkRk4OSRXZQqRnHlE00zG1J3V8qePrTIJpScHPvXI6s4VLI15U0W4ixwDnJqYR9z3rujXUmk9zJqxKUCoTxXE+LtUazjYg4/Gu2y5GFNe8eaTeOGFyyFjtzT28WGZMhzmvGdF31Z3Sg7lObW5JujH86rNI0pyTUezUXp/X4kyfRDWtWftSx6c5PShRv8A1/wBRehZXRfMXlc01fDCt1XFOK5Vt/X3ESs9v6/Eu2nhqOPGVzita30xIuAuBTd+n9fgSpXVmatlDswB2rZtMqBWsdEBt6chZh9MV0mnQ8DI6CvUpK8Tmm7M1IlJArY0yzZ1BHXpWsItyIbOgjtRHGoIxxzUciBpATmvUcNEQX7cAAfNzior+BZAxJA/H6V0NXjZgznb+I7QScDv69qwLlkLfMQK8KvBXsjSF+g2Mqx4NXorZHx24yOetdGDo3eo5Sa0JhFHEcU/cgONw/OvXVo6GOrI7toniC5GO9YN9HGzYB/WlOzWhUXYxb2AKrEdcV5740RJY3BA6cVy1FZG0XfU8O1yMR3b46Z4qmo3DFeXKCbaZnIUQkMOK7jwFpjzTq+M8j+dFG99BJ6HvvhOy8qCP0OB+PFddbo20YAx+teorcuo7F1MnGaezLEu5iPxq42QrHD+OdaENqwRxubgfpXnIcyPknJNeLmU1KSijsoR90u20WMMc1cEIK1OHgkrFO/UrXEWD0xUCQuW6VM4e9dDWiNC1tmPUVb+xlxgVo9R3sWLTSST0zXQ6ZpzDaMVvQpuEtTKo7G1DpqkAYOf8+9WItGBwWXJJ/z3rqsr3MXLTUtpoQGSF+vP/wBeuL8ZeHGlIO0kKetaJaO44T1OY/sEKfu4x604aCowT/n9axlTTdzX2iPkgKFprkkVvP4rsu3UibNMPzDnNXe6ugRE3fNREkHoaF2B7ixsUcMOMV6L4D1cqUUscZxW6fu7lwaaPZNEvhKiNn866S0kDdRWsG7HO32L0bZFShs1ovMQp6ZphQEZFFhETxA8moGj5NKwXIni9ageDIPHvUMRSuLPdnise+0vep+Xk+1PlQ4u3U5690jZnjNZMtm8WTtwK8bG0Gve/r8j18JWTVmypNG7DJ4qs9q2MnvXlt20R6N1YYLEt1U1LHa7ONprnk7s2i+ly7bQhOtSSxq6kAYq4tpK5nU3M+S3K8471NBEX4qpyYr3RJJZlV3Gq0i7QaqOqdyEypIuW5BoFvu5xUx3sdUe5IkAU5xU6xnHStJKyYXuAtt3XirEMATp1rOnUsmDZdh4GD+tE6ArXPUkmZT7mY0WZq3NJtEKqxHNVShzPViqfDc2DsiToAax9QcHJNejKMYxOSmne5QadV4zTBLkk5rF25bo2sxxc465pjSVhLuNRANjk1VurwgbRVxfcajqZN1dM2RmooAztmq5rnRBJIuLC2MmpY0INc8+xorE64I5qdYt3asG1sJou2sAUZq2hxWNV3koohkoww604rzxUuHUyTJUBPU0MoHJq/s2ZK1Y10yuMVn3kAOT3rSmvdugXYx5oWRj1qvI7E4rshZ7mysOh37uhrQjcqvPFTN33CyHLKD94mmtgtnNc0oX1HsTQEDvT5p1ReOtdNFJxuZ1NWRRTMz8mpmuAvJNY1Y6aCWrIJr5TwDUAuiT3xUWeiLUdCOaQsetMXBNdKjpoZjutVrj6mtLWFfUijfDd6kkkKjrQk+vUblqRs+V5qCTAHNTy2epnLyKkkuwnFVJrw+tdVONtieZbMrPcF+tV/Nw3JrfVO5DaQ2S4+U81PpcfmyBzSnflbMK0tDdB2YwaJJjivMSuzkM69Zqyp5SD1ruoLQzbH2eoNA/U11Wi69vcKxp4qHPTaf9fiWrPU7GzuVeIHPWrAbPNfLTaSaYMmjfmtTT23YJIrm5fetcmxtwAbQc1Nv4qJRcCBy9CazNTn4IrbC22YJJs5q6thJIWNRhNvrXoQ02NohnANV53yDXV5l9TJ1FN6kA1gXFkAxzzW9Nu/8AX+RabuULgeUp3HrWW53TevNdtOTbPQou6NK1Q457VaCjqRW01ezN0PGCuRUSy/Pih6PUuMidFO3Jp4x2qJs6YO+wjGmAkmuaTubIV3AAzUT57d60srajQkR2tzXQaFIWdV3daULXt0OTFL3WdL9j3pnrVaS0KnkdPWnKPRHi83cydSJRiTxWZHdFXPNck1Z2RpHVGtpmryQHIYmu60DxALyEKx+Ycdaz+1e5y4mn9o3YJV81H7ZFdlo9wk0QUkGqwahGU4nmVVsy3LYpI2QeacbRQmG610xjyu5LlcwdUjEb5H51DZzMH5NePXSjV5UarbU2bfc6cmh7RnbPr71NanLdBdXJYrNlwRmrkSuvHanSjKEhXJDYrKckVZi09McV3QpKzbIlO+wp08ZyKglhK59qxnFwd4hF66lW4dxGRg815r48iuJYpFXOTWjrz5dDemlzHkc+m3IlJZCDk1atrGUAbs0qk3b/AIc6JSW5egspGIGDitW20s8fLXNu9DNsvJpR4O39KtR6cq4+WtlSS0a/r7ib6FqOzUdqk+yqo6CtlSXKTcVYqdswazaS0Qrlm2X5hWxaITg1orW1E9jf0qMFhk8d66a1QbBivYox/dpnJK9zRsoN8yqcfNjmuo0q2WNQWA4/+tXVho63ZL3L9xcR7QARz2qpvBfryPeuqo7vQGieOcopIPTis+/1YbGAOD061lXr8kNCrI53V9VMcBJavMvEfxAg0q52NNjPqf8A69eLUrr4ma013Ktt8UbZ8Hz1OP8Aa/8Ar1tWXxUtcACcH8f/AK9ehRrxVrBKF0XD8RIJyD5y4+uf61DJ4/t9xPmdO2f/AK9dSrXd0TydCNvHsRBPm9ff/wCvVGbxlCzZ3/rzUyqdRqPQrXPiiKZSA4/P/wCvXFeKdUjlhbDgn1FZ1at1ZmlODWp4z4mmxcnae9VdPcueeay5VKLaM5I2IrQsVOOtenfDuw2+WxXkY/pXNQv7SxDZ7HozCKNAO3/1q34Jc8HsOK9VtDvqXI8gZ71k69qBt1PPGP8AChyurlRs2eV+IL+a+mLMcgHgVnWysz9K+YxNVyrq52LSOht2do8mMnitSCx3AZHWvQhFWMnK2oXGiFlzUUGkFGFaOm0yoy5lcvQaY24YX9auR2OGAx+FVGDBs0bS3TIJ61pQgRnI4rpvbcyd29TTsfm/H/61b1jbLIgA54ropxUtTnlo9S8tsi98/h/9esLxFp0UkTNgHIyP0rotdCu0ee3EIEjcdKgaPPWuTRM32PitvUGmhc8k05u51uwyTpULHHfrVRRMmRkg59qjPPeqsQ1qNJPetzwvqbW10oBwCa1tFLUuEex7V4S1Tz4kbdxgCu9sLgOuc+31raMrpJGco6mnE4z1q0CBWutiGx2SaTpTfYSGMM1GU70m+gEcietQtHxzULsF9CF4jmq81tkUMNDMutNBU5Gc1iX2kZBwAMVlVpxqJpmlOpyu5i3Ol7TytV3sxkZAwK+cxFFxl7v9fgezCrdCrbrjjpUZgUHNcUoKLvbU6oSHiPuAKa0W1STRu0KTsUJny1WLTAG7NaVtHobJe6F3OOhNUHlBatINWsQokTLv6GnRoy1lOVmrGsdiaMc5NWAgI61TnfUq1ncnhhyOalFvjmuO9rktiHK0yZsp71ajezRM3tYqAfPuJ6VtafcqsfJxiuqnC0rkTTcRLvU1XIDAVk3l/wCZ0aumtsKlT1uyiJXY9akUsOc1zQk2aSVmSGRh3oR/Ws5yRCtYV5QR1qncDdzitHqroNijJbGQ+gq1Z2mMZrK72LvoXzAAtRNGT0rOS6suDJ4LUsc1ejtRjrU/FewpzLMds3apTF69qxnTcfeI57rQXytozSh+eRUJyb1J3JVehm3nira5tA2H7fl681VliznJqo+6RcqS2QlB7VVfTQDkDNae0stEXFjvsOwZNVbolMgZzW9P3laxUZalQSsvU4qQXHcnJpOXkaW1HR3WDy3WmveD1FaJJKxLjdhFcjOSaS5u8Dg01G6uQ7p2K0TtK/WrLKyr3rBv37Gl9CEtnvT1Y9K1T94loVpdoyTVWa5DNyelab6sys76EX2lRTXuse9aKKvqPlGNdEj0qvLcEg80qkVfQm1irJNxyapyN1ORW8I2RzSepB5nNQTyZ5BrWKMpaEKFnbbW7pUPlx571GJdomNSV9DQyWNWYLZpB0zXlTdkZhd6ZuTkVy2q2xglwOK6MHVvKxzyfQogle9WrG5aGUHNenNXQR3Oz0jW28oDdwK6WxvhOuc5r5nGUbNtf1+Juo6GjEx61pWBOc5rx7tSJkbUEpCgGrMUm/rTctbSZL8idgdlY2oxnJJNXQT5iU0Y83BNQnpmvWUYtI0RE5461UnbGea3jG+xqZ10wINZF24AJrSKZaZzOqXR3kZqpZsWlya9SEbxOujI3LcjaBUxXPai1jqV0CxsRyeKZ5GXzmp5u46bZbjTgd6GG2sXO7OyLI2fio2Y+tJxVzeKGs3GS1RvMMdaJXbsWlcYJcHOea1tFudk65alG6kY143iz0XTds8S/wC0KlvLNBGTjtW1SLaUl/X5nzUnaVjiPEDeW7D061z/AJx38VxtWep1u6VzTsFZ+cmt7TJ5LSVWRiOea4K1RxfunJVqJ6HdaRd/akQk8nGa6fTbz7OQN+fpVRXvRqI8+TvobsGqLtGWycdc1I2ohjyRzXoqba8zNIz74edz1qkimN89K8zFwalzpFJ30NWzu1UDJ68VpQSq5ByOa66TjUSM3oalvErKODUj2oHOMUVKatoQmxmQnBFOWc+tZwnfRlJaDhLkdaQqJOvehrmRWxDNbqEYnFcX4nso5A3TnitI0+qLpyZwd5pMXmEhRVYaSnUKBTrRurM1u+pLDpyg8ir8NuqjgVz04KPQXqTrDn0pfKANdPIJvUeoApXwRVcvu2YluRcA00vk9a5al7jLdpgkZrZtT3zWieyEzXsJtrKc102mziT8a9mg7wsjkmtTqdItd+z8MfpW3JuijUKa7aUWosi99TOkuHeQEHj+dWrcdMn61nFO71Keol5N5cZw2K5i9uD5hw5OPfGa48bNWsXDzOT8Wa0ILZ/3g6c818w/EXxLJfazIqSNsj468E1wJKbSkaJ8qbOV/ti6ByJXH/AjViHxNfRHP2iTj1YmvQjCnKNpR/P/ADJVRXsa1r48v0TDSt+dD+O9QLEmduaqNNbp/wBfeVddwTx3fnrK3HvTm8d32eZT+FOUUtW/6+8cmnsOX4gXa9ZGP4mql942nuQdz8/Wo9mnK9/y/wAylPTU525uHvZi7nPNXLABCK0uo6I55tHT6RH58ij1r2DwPp3lwoQOgAH6f41hSTdW6HFHfWiGIDnsM1sWk+So57fjXe9Si81wEj64rm9Zf7VIUBO0AZx+FROVohE5rUNEDDgD8Ko22lCJ/p3ryq2Gj7VVGb86tY0ECxrVyyYs4zjGa1pP3rMlvQ3IIlaIj1pYrLdIQVFdvLexmn2Na20cbM9R6/5NUr+2FtyQOf8A61bunYfNrYqRy88Zq7E2cZbiskirGppsu5+vT/61dJZXISPOePb3xW1N8quYT1dic33ze/tVDUyJ4to+9/8AqrWFS7YmcRrFsLZWcYAxkE1gtfL/AHhmsZq0jWK5kfFzjI4pmSO9XurM6hjc96iZDnpVQn3Jb0GlAv1qNwFPByaqK5lqJ3bGAFualtZTFICPWtoxQ47tHp/gTWRtCljXrOi3wdV/rVR+IJaas6O3k3CritxWydzJolBwOtB9TQJDSKQrkUNaXHsNZc1GyZNRbqDIWSoXjzTWgiJ4Qwziqs9krCpaFqZd3pQYHA61i3ulkPgLXPVwynq/6/A6aVZp2ZSNg6g4zVaS0cHGcV8ziKcoVG2tP68j1qdVNCxRNH1qveycEAUqcNLs0um7mUQS/NW7cNtqZO6uzqVrEN3ExzVFkOelVSulqRe7FRWLYq0ke1emaJJJmyeg9YT3qaOMjtWa7Fcy2J1RhS+axO3pWUk3JMHZq49h8uTVG6kx0FbWVzBO7KZuMHmnC/dR1xW0J2ehrykct00jZzTQjPzzRdvRg9BduzrThKD1FTqiJK+ou/PegvilyK7Ykhytv5xTZAMc9apR00ItqRRoC1XoYDjpWDs5Gj0JxETTfJPpzWUrcr7krceuU71YilAPJp01y6lzV0XoXBXNSBN3NaVndJGOz1HGLKc0xY8HmuW1lqCdwkwo61GkneoTs9djS3UGuSaiMhY5JxTlUtJLoLlHJ83TpT1h3HJFaq2iJegydOwxWZdwjBPSt+boiYmTckLzVJr3bnNXGF0dcdtRgvT1J5phmdznNbxpd2JvUsQORzmh5SzYJ4oknFtEbsv6bCpOWq1dMgGBiuWylK5m7uRRfg5ppk960jG2pd+pXnnx3zVR5dxPOa3USFe5CXJNJuJrRxvqPYVuBnPNQSOR9KmyaM5MpzzZJ7VTmkJHWuqnG0UccnqVxMd1IwaVuAa2slqZNl2xsCCGat20iDAKorz8TPmMn7xv6ZoLTAMwrUbS1tl5FeBiK/v8qIctbIp3EQIJxXFeIlCymuzAN85lJGETk0okwRX0ijfQhI6DR5N6AV0enXZhYKa8LGRvJo2Wuh1FjciZQSa17KTaea+cqx5ZCt3NSOUYBzVq2kG4c1hNXldsDRQqV5NZuqmLYcEZr1MJBTTZjd3OanYeYcUzgiu2mrbm6vuQSnFUblxiumMXe6KuZt1JjOBWJqEpwea2jFXuyr2OcuwZJMHmp7K15BrtjpE66Pc1YY8D1qcLQ3d2O1PQQ8DOaQctQ13NoE/QZpj5I61m1ZHRFEJphAA680uXU6ERHoSaifpk0SWpoiM885xU9rdfZ5g2elXytakVFzLlPSPDGqLLbKxYZrZubndGT6inOSa1Z8xVhyzZxPiSB5WJUVkW+lu5GQc15s/dvb+vwOipNOkjasbHygMitFIiRwK8yo3I8mb1NXStRe0+Vjx9a1h4h8vkseOmDWi+DRkxh1L1h4qXOPMH41uWWuR3BA3jPtW+HxXMkmyJwaNmJhLGG65qGeLByBXbiaalC5hHch8xkNXtOvGDgE9a8ylV5J2exrLbQ6WyuwFBLAcdM1cF0rcFh+delKpFox5SrNNuYkVHvOetedCXv3NEhfM2jJqpd6z5PQ4HrXa77jSTZQl8RB1Lb88ev/16wdXvftOTnit6G92XZGBNEWfPrTfIGKKkU02htjTGBT0TNYRi76g2ThcCkIroeogC01hgU1HqO5A+aaFyc1ySeruFi5acEYrXt2wtEHpqLQt20pDj1z6102hzlmUnjp/SvUwsny6nNVV2d1ol0do/u8H69K2Jp0lHXgV6VOfumZWVATxVhEJGcj0qeUDN1NmCHPb/AOtXJardiMEg84rysTL3tTWCPH/ih4gktLKby2+YjA/KvnrUJpJrhmckknJzWdC8pyTKq2SRWAzSlSDXWlc57NjlIB5NSEZHFaJOI3caMg0E5rKo+hb2sMIOaYYSTyKlPTci5NFAcDitCztjkHFTzcxLZ1Xhy3ZruPA7ivcfBFkWt0+gP8q2w8tWaR1sdgLUqhOOaaknkkEmupmvUW6v9sZyetZbSjJY96wqzdylEqXM4bgc1Tdua5aju7lJWITy1XLIEOM+tZxk4yuwn7yOjsRuUZ9OKv26BXFepBXSbOdGoLlYLdQT0Ga5XWdVWaQxq31revNKOgQjfUqQXByOenvV2G5LkDPNcql2NNjc0xyFBPBPPH4VvWu6UDnsCa6oqJi5Pct+WA2fzqERM+cnOKfI07gtVqcn40Vbaycn6DP4V5vJcHJO4+3NFbdG1Nvc+TmQhvU0xom64qFqzqb0EEeOoxSPHxmnJMVtCCQGoGGD0q1q7IGrDCO9KpKmtE3axPkdD4V1N7a5UMxxn869q8Laqs0KfNknFaLpY0nZpWO1sLgSLkGtWFsjmtYsweuhOp4p+Cec1oxWtqKO+KYxqba6gA6etNKjNCGhjIMk1EYs9qltkt2I2j/Komi3fSmMikthg1TuNPV/4aOZiZRuNKB6DrWZcaUdx4NebjMMpq6R1UatlYp3OnsqE4xWLdwdRnB71y1MK1C9v6+46aNf3ioLYE+uKn8vyxXlyhrynpKfQikIcYqrJBg5ok1B2KjcI4wGqwFQLyaiXvI1TY7jgCpUXJ4rKd4xHsTqnHWnJbgtkjminHmdiZOyFnTYnA5rHuySTWkqVmjKD6lJlGaZsBqoXWp062GkEN0NW4XATkVaet2KVmiOWQdaqNcfMRSqK2qCMB8TM3ap1jJHNZud3ZFNdieJM8Cla3ZzVRlq0ZWH21rh8sKvlAowKj2f2mTKTYoA254qNmUnB5rKSaHBX1GOSTTogSelQpW0NHsX4TgCrcTmnJ9zJonWMydaVogqntRKCcTNS1sUpPmbrSFBiudxbtc2uxjbQPaoj854pP3VY0W1yxBFgZNTHjoOK6oxutTnlqyGRM5NZt7G/Iq0nrcFYx7uBhnINZFxbtvPFbUVZpf1+RtsiFo2U1LD05rvdtCOYnEqgZFIreY+T0FZVVfUuCsXobjyY+OtN85pWyTxWDiraDtpdiyMFFVmfNTHmepm2QTE9SKqykit9bXFzDI3OelPeTHI61ooNozm9dCJ3bGaqzSt3quVbENlSaQDkmqzuG4reN7WRyykPt7NpmBxxWpbaaqAFhWFerbQ5pS6IteQqdK0NHhMk4GO9cE22hQfc7zTYgkA4qO+YEHNeDiKbVS7BrUypU4PvXGeJrbDE16OAdqmplNHOiHFIY+a+kUjM2dHBUVv20bO4HrXk4t+82ax7nT6RbuqjINbkB2da+dxCk5XFLctLOR3q5bXPI5rmv0X9fiMtzXpEXBrEu76V8hvyruwkuiISRSYknNKOetepSSsaEFxnBrMucjOa6LJaILmdcttz/M1h3/JOOTW9OP3jTM77MGbOOfWrUKBB7V0paandh1oXrdCwzjinN8pxxQn1Opb2GnHXgUzOTxQtVc3p3RIGwOaY8hNS9WdcERsQBzUTOtCTvobK5A7elQSyHNDd20brzEMoC8moWlyc96auQdT4Q1IqQjHp2zya7yJ2niBJqJ7WR87j1abZBcaYJWyR1pqaPHGMgDNefOm5Sd/6/A86VZ7FaaIQsafbyI3Brz20tDCUhZsLyDVKe8Ze9YxnKTdiVU6MrNqbJwGrQ0jxTLBMAzkgVq4u3Mt0NyurHpPh7xEk0akvwR3Nbv2pJhwQa9KnV56dmY26jGjDcipIYTuyK4J005FJ2VizPePaxZ5rHl8bm3k2vJt+pqq/uWbehcI8y0LVn40S5I+fOR6/wD162rXWIpxncc+/eow6uEo2ZK2oAg5I9sVjapOZPuk5r1KOujElrcythXNRyxswrrTSWpXMVXiweaYycZqHGyBETL3pV45rLYGPB4opp6XFYUDNDLxzVa7CZC6ZNIExWE4rcFcs264IrVtwNlRZ6tAyaMHzK39Hm8tRk9668NJrRmMlc7HSr6MqiZIYiughbzFXnnH5169NKzMWtS3DCT3+pqaRo4I/vL+dbaqN2O3Q5rX9UjUFQRn1zXBaxf5VsnFeBi5e82dVJdDzLxhpp1WNyedwxivKtW8DXCTsYx8vYVz0a65rJlVaXNG5iXWgz2f30P1qm0LA9K7/bXONRa3EMJ64pNhXrVObte5KDaT2oWNt3SsubuXJqxMtsSchTU8dg7D7tZykkZ3L9no8jcla3NP0IsRkZrONW241G53PhXwuDIjGME5HNexeFNHFvAg28gf4V6eFWlzS3LodFJZBV5AFYN9iJjW8lpc1WrsjKnn8w9+OlVpZSeCa4pyuy/IrO2aZjPWs90CFWJiRxWlY2rMwJHcUqcVJ2JlY3bVPJUZ7U6bUkte/wBTXqXtGxikZOo+LIwpRZMn0z/9esOC4M85kY8t71ySqqcuVGyVkaEZJ5FW7Zsckc5rSOjsD1RtadctuCnj3/Kup0uTKDpyuD9eK7KWpzzsnY09pIBzz0pkoEUZY9vetWmSeV/ELWTLci2Q8AZYe/FcTJKeaxlK7OyC0Pl9/lfJ6VNuQqORV+6za2hBIynjv61E+SOtNrS4XXUqyDnrULDcaIvUiV3sMZTjmm85zWqauVoWLScwyK2cYNeoeCNdEmxQ2ccHmmnK+mxVrxPVNJvdyqQeoFdBbTcDmt+pgXUJI5qUdKpy0shPyHUhAxQGo0nFAOaSvYLAQDTSBmhi8hjKCTUZXJNDVgQwxnGajaImoVh7ojaAVWks1JJNDs9wvYp3WnK6kbciud1TR2BY44+lRVgpRsVTl71zHa0MXzbTVa4z0r57Ex9nO7Paoy59SsVIPNBG4gVxSfPqzsQ4W5PNH2VvetKbsiedoljtGzk5qdYyBgmscRLS6KjJMkWPA5NSICBWNCTvdjlqhkpJHNZl3CTnjiuipJp3M4qxQNsc5NSRWxZulTFnRzFg2SKuTiqkyBW+XpRBXdyE2ypOM98UkFtvOSM1dTVWuapmlBaKF96dJFt4HasYx10E3qNijYvmr0Ns0mOtVzrmSCTVmS/YmySAcU8WzqOelOLesehzuSIpkIyBUKwNnkUVNrGkJWRPHb461MltjnvWNtb2CUhSGBqxbZYiluxSelzQjYKOtMuJuDzWkmlozBLW5myyhW4qPzWNYSTOqK6sCCwqSKHnNPk5lZBKVkWEO0YqTG4VstGkzGWmoCMYqndxZNbuKM09TOubLd2rNn0/DUpLls1/X4GyndWKtzp/es6ZfLYgV005NLUmLuyMbyeKmjBUdap2vZmvMSlzjk06GXnrXPe8ht3RKVLd80xk281oodzFsqTHnvUDjcferjvZivoNMR71CeG61sm07GLmJPMAvXmsy4mYHJ6VtTpdzmnVsUpJmbqan0+2M8oz0q52hFs5ZS1OhgtFiUYFSMdorxpS5mShseZHxXS+H7Abg/HFTKyNFE6uNNsYFU7q3Zia4MRT55Jg9ClNauATzXM69YSyA/KcU8OnGpqjOpLQ5i4spIv4TUMVuztyDXvRnpczurHRaJ4du7lQyRnb6mux0jwtKjqZOfoK4MRGU3oPmSR0yaZ5EPIrMu7sW02CfwrysTTs1cUZczsyaK4Eygg1at3YMBmvOlDkd76F7aGirK0fzGs64gDPkV20Ycq5iZblaSLZ1qIgivSp3iNaDCm7rVG/iCjI7VuryXMxXOc1G52Eisp5BJkmumjsUQ8g1KnOM103uelR0RcgbA5NRzMS2elCsjojo7kDzEdDUkPzcsc09tjeD7EzHjrUZ6cdahpHXHRDHwDVeVuKb01ZvBu6K7NtGe9MQZJZjRC3NdhUelyG4BUZFQbsuMHk1bj71kNO6ub/AIcVlukIbHNel6OS6AE5x096xnJbdT53Hu8zSkwFzVOa6VQQTXFVdjy3ojGv5Q7cNVSOYo3WvJ5Xe6OWT7ljz9w61VmUHPrWeqloTczriNiflqOJHDetdUXoNSOk0DWJrYhWY122k648oAJyfrXJKcoVElsdEU2jp7KYzgdR9a04IiCDg1c5v4iZIj1YZtznr2rzHxDpk1xe5XIA4roi1WSiaUfd1Ra0jSriHadzE9+a6qyMsYAY+9ehDDwWqQSldl1bhyepz9acIy4ya6qUERJ2GG3Bo8gYrVx1sRdlS6iAOaqGP1pNJaFKTsMaPFRlcVhUgnsNeYAGnbeKzsN7CqvvSMCK2TVtSOtxhHNIKwkr7DvqTQ53Vo27nbWLbUbDZagBLCtm0XAHatqT05jN7mvp920ci5J4xj9K7TR7sSIpZhkivVw1XozKae5rJcqtZmsaiFi3ZFdFWquQSicTql+0pJLcVyuqSNK/B49a+dr1N7nTAx5bfcTkcVWfR4ZQSy5zXn048srs33RheIPB8VzEwEfB7157qng2a2kLbDt9a7lOz0OerBvVFMeG5CMFTTJPDMirnaa0lVaX9f5HJrr/AF+pVfRHU4Cmn2+hSu33TQ5rlv8A1+QXu9f6/E2bPwyzcla1rfw9GgGV5rPfVr+vuKUf6/plyPSY14A6VtaNofnuMJ/9eojFzkkb8tj0zwr4e2BCUOFwD+ld5ptksS4wMivoqUOSJjfXUsToNuDXI6+4TJHfH9KqTTRcNGYPmE9ajc5NcMo63NbjSCeadFFuIyOtZLViL1tZszDIrVtLUxgkcV0UoK+pjJ3V0T3DNFGMCuJ1/VJ/NMaEgU8Q3GNy6Vr6mPDvYjJOM9zWpaEjHOK5ac03ZG701NyzUOoJ5yKuCPYAc/jXfbS5hdF/Tg29Tnv1zXU6VJtAJzjjv9K6aGi1Mp67G5FIGwSSar6w5FucdxwfyroWrJS1PEfE8mdTlOSTWHKSTk1yyjqdcWfME2SetRbm7mmtDaXcQuS3WhmYAmtrdAstiIZJ5pjLjmpt7wmuxHJ8wqA5rWOliOouTnOa6PwjqxtrpVLYB/WtE3ys0g90ezeGNV86JcvnpjFdtp84ZVy2TitYu+hm7XNWOQEd6n3HjmmiOuo4N60DrTVwA8mjJFLyDqIDSE1VhW1EK5pGXmk2DGkZpGTik1qMaY8VE6ZPNS1diIni7mqlzZCYHI7UNLqG2pjX2ieYTgVkT+HpNxwpryMwwsquq/r8GelQrqJGnh188gUyTQCpztrmjl1qfN/X5GrxfvDBpTIOnSmtZ7Dk1w1qcqL1/r8DphUUhywccdahmiKHJrkm2tjWnoQPLtNTW8iueTg1rQjqazi7Fl4VkXjmqVzZtXRWpOTVl/X3HPGVnqUJbfBOV5FJGmOSfwrkd02jovoMnc4IzVGZ8nGa0LikU5cmpLckEHNape7c0iacEny89accsayd76CtqWIYgOavRHYOmKlUlzXZlNk6zKaHG4VpzcrskYuNncjaH1qN4vQc0m1uy4vuSxQ4AzUuFUVk3djerK1xJjrimwT4NOy6Gij7pZNwccGoppjjnqaxc9dSEtSsT83Jp6rk4qKjbNmTxxDHXpVhQqrW1F2MJ6kJf5j6UqzAd6cpXkNxJFk3LUMuc5IrqTXIY21IJWwKz5Rls4oVnYpKxWuYmZelZFzBg85q5N7IqNrkccGOTSsuAT6ULVjb1K00zDgUW7MWyTVuKvdFq1i9HOEXvmo5ZvMU4q93qQ1rcquCe9Ohh3dRTw8W56mFaSjFsfIiqOlUJkAJ7V6E6HvXR5qquzKc/IqhcRsa2ukjK7kQLbM3arunKYpRkVzVtYsfLfQ3DJhaglmOK8dR1EJDJg7u9dP4cvi77Sacop7mkDsbc70Gam8kP2rCceZ2Jlca1iHPSo5PDC3i8rgZopQs00v6+4zv3KF58PkkGQvWlsfh3AjAtFn1rrUEk1YlSUdjqdN8OQ2cYHlgfpV0WSxD5VxVVIvlSRm5NshuYsoa4rxNG0T7h9a82tFX94pPW5U0jUWJAY10lq4cZ71wTgr2f9fiW+5Lc3BiXNVY71Wb73Nb01bRhe6FklEh600pkZNdPN2EQSKVGelZl/JuBya3UklYL2OW1bcGJrFLshJzXThrJNWLg0OikJ65qzGfU12pJI9KmrIsRknvxRLyOtR5M6Lld4mzU8AK4yaUpPZGlO1iTODmmMT1J60ktUdcBj4PIqtK3P0pyV0bR1K7knqBUe4rzmpi+U1a0IJ5y+VqBMq3at3pr3HblVjofD90v2iPnkEGvTtGdGiU57Vyyit7anzmPg+a5euZVWM81zGpah5ch5rmrU23ZHkyTZQ+1NJz603eQea5nStIzcRROxHcEUnmk9TXPVp8rI5R20NyackGe1c8nYlQZpWVoMjArptHhaJ1JzXM5Wd2dcIu1jvNC2FBnqeP5V0sEAZK9DljOCZg3Z2KWpQLtKnBrnpdJSWXcRW+Docr1K5mTJp6xDIHNKYcV6ap3FcVIvm5qdDitYK17jauRyNims5xWkXcRUuCaqv1rKW9h2GEcVG4xWTWtxiAUpBI4NQ42VgFBpW55NVdWERMOaQDuaya0Akjar9uScc9aySWzHfU07OMlga1Ij5a81tGPu3I3ZLFMfMGCetdJo96yquTyAB/KtaE3zBNdDcF+CnzMD+NY2r3QMWN2fx+ldFWSUbmcYu5y19MSpHX8ayJMyHrmvFrSTd2bLQrTW+eaSKEscelcqnaVmbxd0W0sBKuGHWqd74SjmUvtyK9KjDm3RLZjTeEURy2wcdff9axdW0WK3BJUflV+xjy+8jKUexgnS1dzwKng06OPnHNcLg+b+v8jPluWVRU6CnhS54FdPLZGiSSNXS9Ee4ZSV5zXoHhvwsqshK88cflXoYPD3ak0Zzkeg6TpCxxqMADHP6VfaEwpjFew4WiYXMvULrbGxB445/KuO1aQzy5zx/PpXLNaaGsTNaM/nTShBriaZb7Dljz2q1a25ZgMVMUrks1rS22ckcD3q0vpXZBa6mTY6ZFaIhvwrhPEUQ8/jGBSrq9NmtNdTJjYZwK0bNNzDmuKjE2k9DodMjOOvQD69qvvFla9Br3Tne9i3YqBjnHNdBp7jGT2I/pVwbt7pEnrY2LecINzE5AxmsbxT4jitbRvmGQCBg89q3U0EVdnjupXBu52lJyWOaoOMHrWSd2da0R8xOSTUTDI5px0s0a72YwDafWhmNatrdit1ZHnuajYs1RHXckjYHuKjINdCs9iRpJqazlaGVWzjFaJotOx6d4I10EIpbkcfyr1TRr8PGhJ61UXcmommdFbTD1zVxHzyaoysTZGKXJp3sgT7gfU0gOaQdRG4FJniqvoULnA60hOeTmkvMPIQnFKfXFIGNIzzTWTJ4oEkN8rk0xoQalrsBHJbArwuTUD2Snkikr2GnbYiNoOoWoWsR6ClyK1mEpMhksARyBntVC40jecgVw4rAxqLQ6KOI5Rg0kL0Wql3pTEmuGWXK3n/XkdUcV725my6RIrfdJzUX2OSI8givPng6lF8/9fkd8K8ZdSzA4QgGppQroTXapKUNdyJx1uYl8Cjmqcb5bmuP2T1aRvGSsNufukiqRQNkms5R3SNU7rQieLNNijKmrpp6I0i7ItoxAqaOTGK0lG2wdC7bMT1NWst6VhPRXMZJAjEsc1ZXoKyil1ZEhSuepqMg561NR3SHHUmRTimTfIMmny82ok9TPncMeakhKhR61E735Ub2diQn0qtcSnt2p8iTuwgtSGNyxya0rdNy9KHBaoupZE+0Bcmm7geK0hBRRzajHUEHFQMpB4zWbSu2axfcniJ9DmmzZzzxShKUiJaMgmxjrVRwBya7LbGaZDNKuzJNZF3KGbrW7SjHYqnG7IQSQcGmEHGMZpQS3HIrSREk5psfytg8UX94tbE7kkU1VZQSR1ojF3bM5NbCqhPJqdCqjtXXhLKXvHDiNVoMdS3Sqs0JIORXpyaex5dmnqVJLU1Xlt1QciuaUjeK7FVnVc4FLayAydK556KyGzT3nGarzOTxXCl72pmLCT681u6A2JxUT30NKe56Bp+WiB9q0YI91LS9yJvUuwWuSOK17O3QKMitKULGUncmmgXFQxw4bpiuqUbvUysWRbEjNRvAR2punoK+titcwZUkgVxXi+DCHj/PFebiKUdWWnqcpYMUmrqNLmLEDNePVdmVFF7VRi3JHPFcz9qeOQjOKcXaWjKg+5oWVwztyc1qA/LXZTd1cckV5skGse9iPJrpUOqIk7HMaop3HNYkyc8V3UEt0XR1d2NSUp15q1C27vzXTJ2Vz0oMsxZzx1qzGm4VDTaOhNCmIA9M0NCMZFZrRGlO6IzwKaPm61UfM7Isjf5eetVZSetC6tm8HrqRH3HWopF5qbM2RVdMsc9aY3y9KtO9okt3di5pMzJdJjPWvTdCvR5K564qnT10Z5OYQuWtSvW8vPIxXKXt0ZJSBXNOMuY8SVN7omt+QGY1ayCc1jUh0OaSJRCCOB1oNiT0FcGJhyq6J8iePTHbtVy30tjgkVxqDm7f1+RcdzYs9O8sg7a2bW3KAcUSwt43Op6JG7pFw8LjPSuvsboNEp5yRW2Fpy2ucs1qRX778tis4r82TXrUaaTuyNkO25prwjrXdy66CImTaabU21K6Eb8mmE07q+giCdDjpVVlINTL4hibeKjYVlILjOtKBms+a5VhCtIMmpvrZivdBtzTWBAJoktBaEayHd1rTsyTjmsZNXRVjoNPUbQT3FTXLhRXZJWpmcdyKG6CtnNaMGriEDB5/z71wuuoK9zf2d9yVvErBSA5P4/8A16qXWtmYAljj6/8A164pY/m0RqqSWpmT3vm9GzTI/mpyld3Mppof5eamt7TccmiEE3ZigzUs7I5B9DwKv/Ydw5GRXr0Iu1mEncztS08AHjFef+KYHQtgHjvWlRe4K19jlGkKtjFAYse9edF33J2Vy3a2zTEDua3NL0FncFl613QhsyHI7nQPDwj2/LnNd5o+mCJFOOBjt3r2MPTVtjnm2bO5bdOT261l32qgJjK5rerLlQoow7qRrgEAnb/PpWRc23GTz+NcdRO2hqmUpIcdBUJhyea4mm2O5NDbknFadrb4xkcH3604RE7vYuJH/CO9WVtGEe/b7V1UtWS0Yuq3b23ANchrFz5uWI+Y9KVXblNIMxFkKv3rV0+5wwz19646ScdzWe2h02nXi8ZI5AGfyrVSZCvUZ9K9COq0OezuQm/WCQAGtC212NcHd0Iz7/rSheLLlANU8YrBbHa/zegP/wBeuE1bWZ9RIMjuQCTgmqb1KUUjLYljUTAk04o06WR8yEH2qFuPenGKQdbIjzTGI6kVXLqzRoYAWNGMUKK2J5RjjiomB61cGrWJd0rETfeyRTvetnZWsw7G54Z1V7S5VScLnNeyeFdV82NcnINaKXK9C5aqyO3sLkOAQa14XJHJqtjJrQnU571IDVkpCMeOaTkcmpbGBOTQTSewSAmjdQlcADZOaU880SWoxuc0hqrA2GT3pQAaUvIXQQrnrTSg71G4PsRtEG/CmmEelElYGRPBzUTW+40NdRJdRjW/rVeW0zzSSW40QyWO85IwPpVebTFK8D86xrUITjaxvTrOL1Mm60x1kyOKRbKTGCOa8R4aUJ2Wx6brrlKGoadIwJxWK9m8bHjHvXX9X5IXt/X3HPTr+8MZGxzzUZtiecV5VSCUro9SE7EcloVPTOakSxOMhahT0uzbm0uOa0ZR0psds5fFaKok7Ep9TUtrUoozyamMJbisqiurmbkJ5BQ5xmpY+TzWCXvFbq5P5ZPanLa5bJqKnw6EKViQxhU96y7+XBPNawbUBU1eRlyXHzdeaclztFVTp68zO5qysWEmMnSnvaMwzWUp823QyfusYkJVwMcetaUBVVFUpIVTXYZPNjgVAJcHPeplN30Hy2Q5ZDjmhpeRmkkpJ3C2pLHMqjk1XuLkZOK2oRXLqYyi7mdPcHOck1RvNQKiu9Wvdgo3ZnS30j9zimDdJ2qaslKSsbJWJERsYp3lY5NZJ62M5led1XgdaqruZ+KtJJaha0bsv2toZCKvLpwPavSw1BSXN/X5HlYiu1KwyWw2DpVP7M+7kEUqlLln7v8AX4BGqnHUf8sfBqGSaPOK0dVdDn5bu5UnmXms66kL5xWE5pp2KasjOkRs9KmsYyHzis5S90z16mi7YXmq8r85rnitRW0HwtxW5oY/fLUyt1Lgj0PSuYFrWtV+YcVMFeRMtjVt04zV+3XNdcImOpZ8gt60fZivWtVHUzbJkXC1C/XkVs+wivcAFTmuH8YlRGcmvLxaUYMpM4uE7Jcit/RpP3oya8GppZlI278hrfnuK5e7ttzllNZRjL2jAdp7NHKAxrfR8oOa9Ggr9S5aojkOe9ZmosACT0rrjF2M5bnG61Nh2OayFlEnWu6h8J0UURzIwPFWrEEfWt0vcO2BfRSoyanR+1RFnTFXHZyelOY5HNJrqbQ0I2BPaoHXnkVKi0zeDtqRyCq0nvV+pvFkL+wphG7rSlextF6WIZlK9qrkYPNKK01DQv6NH5lyBjJr0HRrfbGGNbRvujysfIl1R28o1xV9eGK5OfWs6ySehz4elzxaRd0++M3PJrThDE55wa48Q3FXR52Kp+zdjZ020ebHHFbVro5dsla4akXUSOKO5pR6SAvK0qWIRuRWUaHI+Z/1+B1wiXoIR2A4q7DDnFdSSsNmha2wXBrbspjGAM1106KWpzz1ZbYeaPXNQSW5BraNNJ6GdxgTHelK8Vv0BleRcc1A1Zy3KWpE5qMNzk0o73GK/wAw61VkTBq5WERtUUgrnnsNXI9uaeqHrWPK7aDFK8Uxhg05EoB0psg+XrSnqrAQbPmrSsj06VjFXeo7mza3OxcZouLveOtdVWaVO7FBalI3RU9aa16f71fM4ustf6/U7oRGNfHpmm/bGPUmvBr42anZPT+vM7FSVriC5yetXbWbNexgcT7RWb/r7zjrwsi/EMnNamnw+YwFezh0pTscLfKdDZaS7puxn3rTt9J+X5hg+v8Ak169Ki1uKUtLoytZsQiNkYxgn9K8+8S2kbI4yCf/ANVb1oKPuocdrnnt/AqTkDHWo4o8mvMjTvJg2dJoGmGRlLDk44rutJ09YlUbRngV6UI9GZS7nW6LYjK4HU5/lW800drAB0OPzNepTXLAxXvMwtY18IdobPtWF9ukuJdzMSPeuac25G1rI0bdwyAk81FdW3OcjB7d60nHS5EXymdPBk5AyKh8nJ4rhlH3iy5aW/OaviMjpVrRE7Fqzg3Sc8cVpsiGIAAdPWuihZMTucvr9isilsnA/wDrVxGpQYJoxEYu7ZUNjHki2tUkJKnrXHFK5q2aFtesmBuNaEWruFGCc+p//XXUloS9SOS7eRgzN0pr30gzhz+dO19GNFSeZnOT1qs+SeadlsNoYRnkGoycfWmo6hc+YHbaSaiJ5zVM2GZzTWUU3dFX00EX5c0YyKGhJjSuaikXnNONnLQiSuyFhmgnP4VskFuosUpSUMK9F8D66F2qz8njrTcrqxSvy2PWNF1ESqpDGujtp9wHNbR7Gdy/G+RU8KNKwUHk1V7ISWpYNlKc8fd6+1V2ypwc1nfmCz6jSe9Jk+laMS13DPeg880WGlZh0GaA2aSYC5HWmnk1SGHfrS7sVL1Ew3E80Z7mi2gwIzSbR61L2EhpUU0p1pARvHkVG0PrTAb5PqKY1vnrmpQMhltVPUVF9jUDgUcqbu0NydrEFxpquDwM+lZF7ogYHABqKsOeDRVOVndmPPo8kZJK8VnzxNEcEGvCng5Rd3/X4Hq0aybsEMJlb5hU0gWJetcHs9Xc7XK7sRoUc881OkK9hSdm0gbaZahjGOKnWIEVV+5m31JDbAiozbYbio5WnewlNjwmBgmnnCrkmjkUlYTkyjd3RXIFZ88TTDPrRUVvcRrSdtSibNw5LfnTWjxxUJtROpy5ncs2afOPatXK7ccVzU5ct0yaiuyNk61G0hXOKuG9xRYwOX604xfLk1pKKeqKbGMdvAqNy3Woin0D1GNIyjOaiMpOeea6Yq25nKxQu5CDWbcbpWxWsql3YuCsriRwbRk1LHEAc1pTgmtCb2JMDoBTltZZ+ADVKi7pRRjUqKMbseNDZzkjNKNAIbkV2Rwel3/X4HBLGF+003yu2MVM8G05r0KUVBWPOqT5pXInTcelQy24VCQAKirEpXSMG9icSE1UaNzXiVakoSsaRldFWWNsnrUDR881UZ3RTZE8YJAAyant7fZzjmm27WIbew+UgVSlcgmiCfUWth9s+SK6LRWO8cUqq1u2aR1ep6BorZiGfTity160oJvUme5sW2dtXrZfmFdUTnbNSO3ym40hjBNdSVjFgyYXAqrIoo5eoJFG64U1wfjNiB+NeVjldM1hvqcWj4kNa+l3fkSA5rxZqxdrmje6tmLANZA1He+GFTGP7znf9fiNLSxIs6iQEGta1vAyAE114d6u/wDX4g0SSy5UkVjalP8AI1dLnoRqcnq37wmsdR5UnrXTh5O1jei0pWLS7ZAM4zVq3VV6V1tWjod8E9y4i7wOKURkNUWNk7OxLGoPWpDFileS3NE7DcYNQzDOTQnbU2iys4yO9VnXrmtNbG8WRkYHemBVzyaJ33NE30GTKGHBqlIdppRTa1QKV9DX8MxFrgPXoen7Y7frVczSPNxurM7Xb0JEQGribkNdzkgZFZTs9TfCR5IXNfRrBht4PNdlpmjmZRxzXn125+6jzcdZu502n6IYlBYc1tWtoqgYAqqdJpq55aj1RZa0BGBVeSz70VYWeq/r7jbmGpFtPIq1Chz0rDW+mxTZp2qk44q9HEwINejRldGEnqWonIHNWF+cc4rritDN9yOVMHiomJqUujEvMglX1qrIDWUtxogYHNNK0K9tB3DYcVDKh9Kpq4k0yDGTTXXFZSVx3sRlcU4ECs0GopNRsM0muoCfWhulOSTiBCVIOatWmWbArngnzWQjUETom7mq80xyQaeMcowsaU9Sq8nvUZlNfIYmbV7nfS3G7snrThz3rwZb3PQitCRYifmzVm2LKRXqZdJxejOSutLGxatlRnrXQaE6eYM9j/hX1uDkvaJnj1FodxbTQxWwYsACOuazb/xIkDEIRwe/f9a96rXUFaIoU7q7OZ1rXjMGw3U1w2s3bshxkknFcdTEu+psoO1jl5dMlkfcRkmrGn6JKZgWXjIx71lRk+bvciasjt9D0sW8auw57Zro7GAsw475H6V6EHeSuYSWmhuwXSWkI5AYd6w9Z8VYykbAse3f+dddetyK1x04XMaOd7l97nJbqauQLjqcCsYtvYb3saVqTgc8VbMZf3zXSnzKxmyvNanPFR/ZAoIxXM4voVclgi2mr8NmWAJ6GnZNXFfuTqrRcKx4pfPx948j3q6CSY5O5j6vJuiI6f5FcRqYAcitaiurFQMadQPxqIH1NclrbGi11Hq+KnSbjmuiMtNRco/zvemmfnOaerCwxpc0wvnrQgsGQRUb4zVq4at3Pl88nBFNaPnjijmsdEV1EeHjIqBw3TBpwlzIGMCnvUmeM1Um3sCEbmon96UV1FuRMmQTUbkjAx1rWCJWug0DB6ZrU0TUns7lT2zWltd7BGzZ694S14TRoC3pnmu/0253kHceK0joibG1A+cc1esro29wko5KkHmndtCj8Wp1N/LaTFLq1IIlUb1HY4rnbqPDNj1rGka1o2ZWHPFBbHWuh72MHuJn3pCSe1JX3Yw3cc0maVuwtxfrSAmi7AM85NLnNMdhOlKSTQU0Jk0ZxSfYVgPvR2pdNBiGk25oYMayZ700rU2EMeMHmozHg5qulgQhj9qie3DDkDmlYRVnsFfOVyayNQ0NX5AwaVSCnG1i4T5Xe5kXOlyQn5VJrOezmkc7gSK+fxVF09kexQqpq7FS1aI8g/lUvIFcDWvMbyd1dCxzFWwM1cgkz1NNS6DdrFkTAD60x5hnmictLkKJCbnPCmnrl1JNFKaeg5qyKN1D8+aVMBcUnH322HNsRyquCcVl3H36qaSVzWk3cltWPUVpxLkDdzXDON27Gs2PlACVVbHetIQSdiEClVGTT2kXbWzhdaBK+5UluFV8UqyKy5pyik0ola2uRSuMVAPmNU42SC3Vkc1o0hz1JoTQpZOdpH4VvRoe1d1/X4GU66gtSYeH5OnIpw8PuD0au+GEne1v6+44Z4xX0LUHh0AgsvNXk0tIh0rshh+V6r+vuOGtXc9LitbqO1HkD0FdLS2sYIR4cdaglgrKVlsWo6Fc2+Dmo5IdwwaibKi2Z13ZKc8Vlz2hXOBXl4igvi/r8gTZQuItvaqMic81xwvF6mid0Pgtt5zjNTsmxemKbneVmJ9zPuWOTVRssfmrtpxbiTuS2/ytxmui0VvmHFY4iN1qXBa6ndaI5AHNdHa9Qaim/dRM7GxamtC1yHB712w8zBmvG+VHfikKHPINde+plsRyAgHNU5WxxUzvbQNypdE7Dn0rgvFo3HHpXk5hblNKe5xMymOQnFWIJDtrxp6otjpHZx3qsdwbNKIuo5ZyPwq7a35UjmtqbtI0sXDebl+9WbfS7s811uzVyVuYN7yTWJOSJK3wqZUNGTRblANXLcuxzivRcu56FOatqakTbVpWk9OtZ6WNIasVJMdaeZs96m1zSwgY/nTJI93TvQ7XVjaLIpIMDg4qrJHzVxuzWErkTDFVnO3PvVcuhtG5DJJgVVkOSauS00Bbm54bnWMjOBz+ddedRRIQR6VnU+G6f9fecWIg3O5iahP55I65pmn6bvbOK4qk9NRuXLCx0umaeI8ZFdrocCBVzjjms6LjzHk4huR0axJ5YxjikCgNXXKCvzI543ZZjTI5p3kAjpU1FfcCJrUZ6VPb2e88fyrnp0U3qEmadrYkYOP1rSjs+BXXGm1sZSeoNabe2KQL5Z5raCZmwxuFQvHinNDIZEyKrSx81nKI13IWTJpFjyamKuF0SLFUFzEADV8qRPUosmDzUb8CsJOxaVyNj+dM3c81mUhQ1HapewmBFIQaq2lhDG6VY04gS5PSoiuWaYnexttInk/hWHdT7pCajMZJQNaKtuQ789aYzc18ZjZWWp300IHp6yE/SvFkkehHRFiKXA5qQT7nGK0pVnTat/X4nNVVzWtJsIDmr1lfvbvkHPNfVUa3KuZM8icbvU3Y9ed4du/gDjmse+1BmYksTj3r0/bOaTuaQiZM900rck1Gtr5zcjrVuW1y2rGhZ6HHIo3evNXk0OKI5AruhHlSSMJsuw2gOAB6flV1dtunHWu+lHqzn3MPXNeYZiiJz0+lYMW93LMSSe5rlrSvU1NY6aGxYxnArSgizjiuilsYyu2adtBxnH61dROMCutbWIbHNDx0zURhHpzWVRqwK17DorbJB7VfRQB+FTC6VmURTuQc9az7m7CA5JxWkVroMw9S1JQpy1cnf3YZjyMVU5Xui4x1Mi4kLHJqEufWsdEaJWFElPWXjPNOO+gPQeJT60jSnFWn1YDRLTlbJ5qu7E0S7sU0DfIB6mtIrW4tkfL0bepOacee9ZSTvc6FIXI70mwGphe9xyehXmTaetRE571s21rYjm0GnmjGec1bV0K+gxl5qNo+c0Ju1hXsNIpAdrjmtYX6k3tsdh4Q137PIFZsV674f1VbiNDuOK21uU+6Ous7neoGe1aEcn500Q0aGn3rRuELcHjrVy6iUDOPl7HNRLSWho3dGa64amkZrS5m33EORSDB6mi+ggIGOTR92lqNbCHrmjOKfkAZz1ozQ0Ugz60ZzSFoJ070ZzTt1DfUXOaQHmkNC59DRn3os7hcDzzTdvPrRsK4jLzTSvekDEK0hTdT2HsIYs1E9ur9qlN9BWIJdMWTqM1VfQYyxO0H3xWdSCluXGo47FSfw6H5AFUpvDzj+Hj1FefXwUGtFY7KeIbKL6HJGx4OaZ9hljPIOK8+thJQ1X9fgdUKye44qyj5garTFj0rilfl1N4vqiOJGLZ5q8AVjwKmjBp6lVpJlaUHnNV2bbXQo6mV+xXnuQARk81TIUnOazrptJI6aeiHQyBWAHNXxcKqdaUafUctxhuCR1+tMebNX7PmYmytNdbDgHpUZvDJ36Vpy2WppbQhd9zbiamTey8E4rNfFqF7LUl8h8fdNOjtHY8qeaJSZDqI2NI0Uu26ReO1bq6ZGozivcwNHlhd7s8XF1nKVkDWSD+HFN+yL2AruWiOOzGNb47c1DJDwfWkn1BLQrtBzSNHt7c1LY7EMik84qFkOTms2k0ax0VkQyKOoqFxis5X0RUVoV5EDjmqNxbcGuepFNWYmZF7a4zxWTNCQxBFcsqS/r/hgTsWbOE7c4pt4AoNcXK3Ow2ZEgLMSajK816aTirEodGmDnNb2icsOawxM2lY0g7u52+isRgZrqbTPFYU3dXJmjXtugrQtySQBXdFaI53ubNlDuUE/pV1ogBmu6CbWphLfQo3Ixn1rNnbBrOSswWuhSvJPkNcN4mGc+teVj17rNYaM42/AXJqtbTYOCa8eEbwubbovKd3NMkTNYq9yLleRdv41HvKtnNdFO97oqD1JVuyveoZbndxXTzuxo1ZlGchgazmgy+e1b4efK9RK5MqIF5yas26KCP5V2xbaudkS1nnApjuUPWhdmdERyknmpAOKpbl8woODnNSBveiSuWnoJJjFUpcc81UO5cHYrv0qrKDmqV7nRF3Kk7VVYgnuTRrsaRVi1ZPLAQQDitm31CSYAEmlOXu/wBf5mdTlepft4zKQT0rWtI9mOMCvMqqzOGpI2LSXBFb2nXxi281lFuLOKUDct9Sd0ALfhVuK55yTXZTq31f9fiYctti1FdjvVuKZWNdDtLUlxZZSMOfWr1nbAEHvShDW5lJu1jZtbbgHFXlg9ea6JR1OaT1Elt81SuINo6UutmPUrBMU5owRmjfQT3K0q4zgVUkU1E1oNPUrupzQi4rOn5lkh+UVBMcitZbCWhRlUZqu9csymupWkPzd6jySayd1qUSRgnrUgBpx11E3cOaMc1oo3IuI6dzRERGevWs9ExrUsyXfyY3VQdtzZrzsbU53Y3p6DeTSNXzeLi2mzspMaOetSJwea8OR6HQkLgDJNIsw3DmlGOqOWpLQ0rW5zgA9avRMTzXuQr3srnmyRbinI70skDTDOa9TDydQqnKxWNkRzirdnZlmGQcfzr04LmaRUr2Oh07Tn2gjjNXDY9CT0r2KVOXU4ptEMuyAHoKw9T1QkbIzg+tdDkoxsSlrcxDGWbLHJ7+9SwwjPNcbinK7NNjWsoj8pHY1rW0O4g9u+a6aasZTd3oaES7B+FSq1dFzMeWz70Im5h71nU1sikupZjTyx7kc0FsdxTa10GVruUIuW61y+rajtB2t+Gau3u3HG9zkL7UJJZG+fjPrVCSQu2SaytudDS6FW4bmoi3FTbYBN+3rTg+72FaWsDfUN5Hems565qubSwhynPepo+Oc04N9QbJN35VY0+HzrqMf7Q/nXRTWpEmfKwBU9eacAQaxldK6N76WEyVPNO8z+9TjaSuD2IpAGOahKjk5pybuhJWExml2kHirUrXGkPaPC5qrKMHFKDu7hdXIyDk03ywDuPNaKd9iGiayuGtpg2eM16f4M8Qb1TLDj171tzWtzDjZo9O0e+Ekatnr1roIJtwBzVX1Fa7sy1G2ORV+1lMy7Cc46Zpy2uSmxJoup9KrnP41MRaX1EPApAKtAIM5oJ9aJPYaA8Ug5oEBHfOKD9aL3HcBmk6Gi/YW7F696MYpX6Dv0DpSE80XGGT3ozRzBYUGkyfWk2NiA0tULqB4pBStcQmKXb3xQ0U2KBRtFRJdBAYQ3NNe1Xqal32BX2K8tkjZwOKqy6WjdFxSlG6s0XGTRSk0LexJ61Vk8On0rz6mCjNbHXHEOLsPi8O4wcHP0qf+wRt4/lmpjl6/q3+QTxDZXudAyOxrKu9DcAkLRXwWl4/1+A6eI6GLe6LcKeVNZ0ttIgIIIrzsRTdNpo9KlVUojEjZWBJIqwrHbzzWMZOTsbXvqIWYU6O3nmGQpNbQi1oiJzUVdkc2nzckpVRoJd2Apq3TlFvmCFWLJYLCaVgArV0mj+G2kXdICa3oYdzmr7GGJxKjGyNr/hH0UcqPwpY9ERWHyjFdtTBUdLRPN+syehoQ2iQj5VFPZRXYkoqyOV3k7sidfWomXinbQpEL89aidc0rWVwsRMAvWoZAGrOw0nuQuMVA6g5yaiXcq3VEDjHFQyLms5X3ZoivIDVaQetZSvuNso3UQbNY1zCPMrC2ommSQIAvWqeoDrXD/y9E9rGWy9ajYYrvbErAuQeTW5oRy4rnrtS3LidtpOQwIHNdTYseM1nRaRMzYtzwOa0LM5cDiuyMjmaN6zwq9c1YeQ4rvjK6MtDPunz161mz8tWVQImfeH5DzXHeIecnNeVj7yi4msTkNRj+Un0rJhkxKc15dKNotGttNDSicFQTUuAQTWFhW0IZVABOc1QkbB5rpoq2rCA0Fj0prQux4pt2epo2Ma0ciqk0LJkVvSd5IlS1K/zFsA1bgBHOTXpJ2VkdcJaFuMlqJkJGaW5pzakSsVPNTpIp4NWnY0vqKSBRvx0pX7Gq1EeXK8mqrH5qqLeyLi7DSMmoZkyKqOjN4sozQnk4qOG0LtmiU9TVy0Na000uMnmtK20wJjC4rgxNXXQ46lQ0IrcRADHNWo+wrmcubc57l+0zmtmzQuQawc7GU3Y3rGAkDNaKQGuiMtDlb1HiNgeKuWyPXVTd1Zg5Gpbqy4JFX7eQqQTXVG+xk9djWs7jgZJq/HcA9T+NdK1RhJDmnUg45qrK3U1m0r3FbQpyHB4NND+tO6WhJFJiq8ijOal7DW5UlIGajVvesE9SgZvSonrRz1sBUnNVnPNYPV3KIJVzzTETnJrFrqDJlGKXNEdybCE0gOK1TENeYDqaqtc89a5a81HVFxQLOW704HJ615FafM7dTZD1TPOeaGTNcGIpXTRvTlZjfL9qUqR1r56dGcXax1+1VhjbjTo4Cea6MPhufc4q9XsXbUbG5NaduSe9XBPnt5nOnpqWosZ5rVs3Qrz1PFe/gppScSVoy3HY+d2zWvpuiY2uQBketfRYWhzPUmrU6I1ktkiULjmq94NiHFew1yo51fqcnq965k2qcY5JrGfcepzXn1Z8z0OjSwgTNWbeDccYz+NZpczE9jYs7fp2GP8K0oY9oHp1rsguVGD1ZMPrS7ueTTnPUe48Pz1qeGQZ5Oe9KMuZ3Hy6EpmG3OarzXaquciqi02FnYw9Y1UKDhu3r9K4/UL9nZjzVVHskaQS6mPI2TjpUTNzWS6miRXnYmoCTmhu2g0JnuaXfiqjroU0G6kLGm30ZPKPT3qZGrSFyetiQEscA10PhrTvNvISRnLqP1FdlCN3oYVmkj5GkUZyOlOii3DmuKUnqmdaWhHNHtNQOrA57U6c/dCXYTNNJ5NbeYdBCp6inhc9azbuw2JSyiPmqLruJNaK12Q1rcPL2rn1qFwFraC3M5tkLNg9a2PDmtNZXCgsQufWtrXLpux7F4U1sTxr84Kmu60283AZ6YoT5k2U97mtDKTyTVqCZo3DA8g1S1Jehof66MMO/Wqs8bI2M9aiL1JaIzzwaQdKsEwyD3oIJ70n5j2G496MYoJfYQn3ozRsUlcAeaM800gWoE4NGcUIAzzSEmiVh6B3yTRnvSewXEyaN1F0Au78qUHvQ2FhpoJ5zTQai54pQeKLAL3ozUpahYeOKOp5qRoNuKaVzS33B6i+WB2pDEOw5osgG+SBzikKYNKwIY8amoJLRZO1CbB+ZTutISQH5Qc1lz+GUYkhRWNWhCsveN6VdxRn3HhQk/d/IVHH4UJYEhvpjiuZ5dH7P6HUsXoXIPCYXBYCr8Ogxp0AralhI0tepz1sQ5aBNoMbjJVaq/8IxDuyQK6nCLd2jKOIaLNvoEUbA4HHtWlDbLGvyihR5WxTnzjnjz1pjLtoaTIuNIJpjDFPoO5C/Jpj9KnVjuQsveoXXmlfuO5C4zURAJzUuN9RkEh7VE61DGtiBhioZPas2rmiK7g1WmBOaxkNLW5SnX5eayrlPmNZvccrlcMV7VVuiSSTxWDhrchu5RYc9ahPJrR7XFYaAc1veHwWcVhWty67lRVztNOyrCuksnJUYrno7aikjYtie9all94CvQh0OeTN+1TgGrJiJGc13xRzyM+8XGay5TyaymtSo7GffEBTXI60C27PWvJzBe62ax3OU1GP5GrnCxSc59a8/DxclZGpcjuyBStfNjrVRw/VjsIJ3l96VbV5Dk03aAWtqW7bTWJ6da0I9GLDO2slFzd7f19xDmMudN2DIFYeoQBckiuiFNpr+v0HF3MoEB+auWyB/avRjtc6krK5NJ8g44poJbrTWxpGSFMOTTdmKWpopCBiDzS7801udFxANwzTSPaqB6u4mRTHwRgUR7myIfIMrACtO20RioYJn3rGrLl1IqVOVGha2RjOG4rQitQAM15k5c0rs45TFaAKaVIznOK5+b3rJj5upbtoyDW5p/bJp3vIiqzfs3AA5rSicMK6adnocj3JE+9WjYop610Upe8TLQ1ERQAanhRWPBrui9TPWxdiiIAINTbmWt3LsRe44Mzc80yQNzgmk02hehAY2JpBGTSirky0IZvkNQHL1LBbXIJoCTmq4iYHJ4rNrW6GmDCo3Q9aSWoIpzxnnFVhGSanZ2KuEi+tQng1i466AthPNxSb+ameiuCQNJgZzVWe8EYPNDnZXHy3KE+pY/iqIXm48nmvLr11JNf1+Zqo2Jo7r5utWY7jd0IrlU431/r8RliOb1NP8wVjUlbcRJGQamEIbmuVxjrIUpsjeNV603z0TisFUjTT0M7OQR3AZ+uK1LW4GBk1hSV53/r8zRqy0Liy1dsZ/3gBNelTfLVijJ6I7DRYUlVc/XP5VvRKqrngZr9Bw0EoJnMveDyy3OKragqIhzjp/hU1p+9Y0SOE1BN1wxqn5Oa4pr3ro1JEt89RWlZWu4r7U6UW2TJ6G3a2uxQefpUkoA6mutrQwTIHfaOuKgN2N+MispmkIrcnikyAc5NEt0IR97FOKdtBN3M2415Y+N+PfP/ANeqM/iAbSN+c+//ANerc9DVQbMLUNT84ZL5/lWTPPvpNtq5XkypI5FRNJ2o0auirEMhz3qPoetTvqAYJqMkg1UNtBoUNSFuam12CY9WzUwbiuiN9iS7YWxmkXA6mvRPBGjmS6gZlwAy/wAxXp4WJwYmTR8NqwJ56VPFKAMGvL5dbnq9CK4lVjwcVAW3HGafLboSxuFyeaOKXvW1Bu7H+Wu3IpgcKa0toTdvQRwXGT3qEoc5PSnBkOQ2STavvVV2JzXXGxm9iu2B160ROwcY61re7sVCWtzvPB2vNCVR37gV67oGrCVFJfPtWSsnYuT0udVZzh/4s1fjfPWru76ietmaGm3YjlCvyrHB9hWjNagqcAFT0rOb5ZXLtzIyp4GhchhiojkdTxWildGVtRM0ZOKYADQ2aGLdjc80E4GaT8ythAc96M0wQpPFJkmiwCq3HWkzSYB160dPpSvrYdtRCeaDzRYTAGnc9+lDH5CbuKO9A7MOvJpd3FMVwzSg+tIodn3pM0eYhc+9KOaTC4e+aUNSH0HbgTzTWGeaSBjSnemkYoE9RCAaYyKeadkFmMMAJ5HFIIFB4AoW90LyH+WOc0ixAZNKS6jQ2RRUZUD61S1QrahsxzRnFAxrHPWmMMmi2gMNoAzmoWbrRbQNyIgdajfFS9EPUgf2qJh61LSsO5E4qKQYqL9xoryKQeKicEDmk2i73IXFV5ODWctHZFIqS5LdaicZ6nFYttml7lSdBjrVCeHf0rNvuN6lZ7c1Qu1xnNYddCeW2pQdeahYfMasm+lyM8Gt3w8fmHPNY1fg1Kgro7OxbBHNdHYt8oxXLRZE0bFqc49a19P++M/jXpw1OaR0VomFH0qyzALzXoLa5huZV++cnqayZnwSa5qm5a2MnULgYODXN6kd+e9eTjnzKyNoI53UYvkORXMXFuRMcCubBRs3/X6FXuyWG0d8datR6SW5NVXqW0SByLcGmLH2zV2GzUdq5n72gtWaVjYKTkitQWiKnNdtGCUbmczO1GFdhxiuM1392xAqZx95cv8AX4F030Zz3m/va0YZP3ec13RStc6U7Iikuizbas25JHNaNqxrGxYC+tRt8tZXuapkbsMVA0nzU7O5vEljbcKc2SKJMbaIWRse9KkbY5BqfaW0KUy1bQZIOK6Gw2+WBmuVzeqZjVd0SThQ2RjNEbEDJrz20m7HOk7DmG761LGnHTik7I1vdFqBK0rclKzbaehEi/DdlO9aFrqGRzzWlOrbX+vzMZQNCC53kZ6VqWUuOc12U5JPcylE2Ldi61ZtYHL8V1q83oYt2VjZtoCE+bnilmTaB0rrfQz6iwAE4P6Va+zAj3rpsnFEyuVbiAKTmqMrhM1zXUWCVyhcTZbGaSPpnvWd9bobQ5xVeVc0S20EQlfWmFTjrUrQa3IZYC68Cq0sOwU5PS4dSq/GahkrJ2eqKIGbHU00yqoySKwle12NblO81FUHDAViXmq7nIz1riq1mo2NIJ9SqbotzmnR3B6k149Wo0/6/wAzVItRznjmrUM5HOaxqV1t/X5itYtpcHrmpVuT61nWr6N/1+YupPBOS2avJcfJ1rhhiLXv/X4kyWuhUnuCTyarM5J6mueU5S95j5Uh8bHNaNjKc9a2o1WmDNaInbk1NDLtkBzXa6l5qRizq9A1TaFyen/1q6u0n+0YOc/WvuMJiXKkkjFR5S5uVVySPSue13URyqnkjv8AhXQ31Gm0czJEXcknOaVLbJAwKzauVdssQ2mea0rK3CjIxxW1KOpk3pYvbhGnJxis67vQuTuA/Gt5+QrGPc6uScZY46HP/wBeo7e7MjAsxx15rk53zWZq1obFvPmMc8YzWXrd6yRnDEDr/KuqDVhROTmuZJH+8cemajaZgM5I/GoskzZeZBLKcYBqtI/PWkvILakTMSfWmsKeoyN6jzmnyhfqO6LUXFNLQENK55puD60k1sNEidasQLvcAV0U46kSXU6vw7pbTSrgcZ6/lXrfhjTRp9mbk4AhjMhP0ANerh1aLZ51bWSR+cg4p3WvFT1PX1ehA4O6nKOOetXKSewm0GDnBoIwc0m9LjastBGb3qMg5zWibejEkSAgdaZPgDNXGOplMozSnt0qBpO9dKitSL6ERJdulPiG361SsmJSSNGwuXtpVZWOa9N8HeIt6KrMM9Kzk00mbLbQ9K0rUvMjVsit+2nEig55q+nMiWy1HJjBFdHol6t3bmylIDn7j/0qayvG5dF68pFfWpdyhUmRTjgdayZI2RirDBopy0JmtSPv1oJ5rSxFwzzihs/WgGN5oNLcYgHNL702wuGT60n40k+okKelJnFCdyhCaM5FOwXDHNBzSQmxRxQW7UWuOwE0DOc0DuKDk5ox3pBsBPNBOTzRYGKGyaUHmh7BcNwPWl3VNmFuomc0m40x3HZo3UnrsK/QBJxTXbIoe4MiaTFKGzRfUrYXdxS9OapNXJYpbIprN70mBE2TSHk0x36iHigH3oewhjLimtkGkmAx84qJhnqaTYX6kTjFRtxSeqHciaoiOuTQMjYYqF+aykuo0yJ+B61A3fmsmWV3BqBxQ0UmQOnzZqCbB7VjLQuK1K0o3VSl4JrB6IteZBIM81n3cfUmslqxPaxmSqQTVZ075rS3VE81iMjHNa2hSkSjnvWc3dO5UVqdrZHO010WnNlRzXJTdtCJm1aP271rWUm1xk16NLQ5Zo6OwmDLU87HbwRXetjmZk3j9c9qxrybbnmuertcuKMO9myetZVz82a8evrqzpiZN5b+YDWY2lgvnFYU04/1/wAAGyeOzVB0qdYgB0rCV3cEBG2nRHDcmnF2kNMvw3ITv0p8mpDaea61VUY6f1+IpRM67vy6nNcnrhaVyRUc+pMbXOeeNg54qWGcjivToyXKbRdyzAgeTcSK0o1AFErWNYMkTimT4wazbTZqnqUZztBqsuSc5ocn3NlLQnEhC+lTW5LNzRKSsHMjThtFcZxzQ9kEGcdK5Ksre6SpO463hweozV+JSvTisJza2FJkuCeTUiJnk1yuVm2TcsJHkVYjiOKyaTV2F7FmFMe1Wohmp1SsgexPHH3q1AhB61PM1YlmhCxXvWlZ3JXGTXTSqO5jJXRv6dOCRyOfWt+xZGwx28c9a9jD1Lq5y1EaPnKiVWubkEdR+ddLkm7kRTRFb3exuTxmtOG5V0BJxW0Zpx0KlEhu2BX9Kx7nJJ5rkqTs7kxWlim6c80obaKlO5SYu4k0jkVpLsS9SvI3pUYOTU21AmRVI5qvdxjDYFact1qK+plTqQapzvtHNc0rK9hoy7vUFi7isTUPECxAjdk+xrjqVeW6uaxiYd1rkk5IDnmo47ksdxOTXm1ZWtb+vxNeti0kue9WIj3FedWvfRj8i5AOeauwLXBU1Yrk/IFKpOa55NtiZahbtVoOduM1i935gQSZzzSBaLPdgxc7TV6wbJH1qqbtLVieiNqP7n4UgfDcV6lNxk0zM19KuCrjnrgV22jXYEalmHbPPXpX0+WzXLyoxluWL/VliUfP+tc1cSNcyb3r1bt7jY1Ys471Yit8jPp2qkS2WIYwvJFSm4SMda6EupnuzPvdVCpywrAvNY3Mfm6f59amUm0axRnfbPMf5m/I1dtpslcHP+FYuNpJgzatJiUHsOlZ2tEMh9zW/XQcbI5lzj8KgkfPWk0kjS9iB3yetRNSe9kA08c01jxwap6DIzn1pmcU20IazZ601SOpoT0GIT3qPLZ60oqz1Ha2pIiliOetbej2JkccdxXbRV2Y1JaXPSvBuh72TcCAMdvpXSfEHWYvDfg688tgshtn799vFeulZI83eTsfnV97k0qLlq+fejse49BsqkEnNRqT3qlFE6XuPI4qNySOtCKdhgzTjuIyK1hZ6mctHoMdyg5NQySk96203M276FWVyc9KgHzVotE2yWkhRxzUqDcw5ptq1xRSuW1XaPetHRdVewuB83GayTubUrao9Z8K68J40DSDpXeadd7kUhq0i7qyJdjVhk3YJq5DMUcMDgg5q/Um9mdbohTWbZl34uoxkZP3hWTrkMazEKfmH3hXLGXv8qN6sfdUjJIpCK6r6mAgzmlOaGhID600/SlsWhOtB4FD31E97ic45o96G9RDsk03JzRoVdASc0mf1o6Cb6AWwaOTRYNAo5ouO9gANLnBxQtdxsUHB5ozStdj3EoxzQ20SKeKAaV+41oL3oyabYINxzR1NIGHNIzcUW1BK4DNB60PUHoiMjLVKqDFJoG9BHHvTCfWhaAgDH1prHmn1BiDPWg5HJNGghucnmkziltoMQnimc9aQDG61GRTewkRuuetRstLyKbImBqIjPNFhkbVCw7Vi9CkQy4U9agdc81LGQyAj3qGQGoaLiQSDFVpFrGaurlJ9Su6HuKqTRHOKxkaJrcgZCOtUrpNwzXPbsF9bmVcRYOKqyp2rVN2J3ZAVOau6QxS4FDSs2CWp3WmuGiU1v2DcCuKGjaJmtDZtZCK07aTBGTXdTnoc0lrc39Mn/LFaEsmI816EX7pzvcxtQnCA5rnL67LMea48VOyNIRMmZixNVpge9eZLU2RXaLd1qKSFQKjlKK0i4qPdtGTWTSTaAiaQZp6etZSdtRCtLio2lOOtCk1GwWK8mT3qpc2QmHIqb3RMrdDPn0XIPFZlxpnlPwOa6qGIcf6/wCAEZdv6/EjWJ04HFWrdmB5JrsjWubxkiwzkCoZZGxVRZpfUrud/BFIISTmpl8RpeyJYoS7YNXYbUIwNZzm7WFzM07fAUADmnTRFhXBVq2ZqtRsMRU1djjyMmoltuZsk8vNSJHgVnLYObSxYijPGasxoayTurMTaLEaY61ZiQ9alsSZZjSrMSY5NSnrYTLEYq1CxHOapTaaQWNGyuijDJrctdT2Ly/P1ruw9Zr4mRKnzbFhtZyuQ3T1NVn1jc33gc+9dNTEpLRkRp2eo+O+JIOefrV+21Nhg7v1raFdXCcbl1bzzByc5qCZAeRVTd1dHNbUquuOTUJIJ61lztMBR1zTX9a6eYkgkqPOBVpXAUSkdDTJHLg81d9AsjLvp0i5YisDUdXjwcMPzrlq6WHFHHa1rRdiqH6nNYU1y8mcsa8qtJudkbR7EcZOc1egJPeuWrd6srrctIT0zV626Vx1VaN0XoaFuOeTV6NdteXUk76GbQ85Jp6Ak1zsL6WLMYIqcHioctBoY45yaaTily2V3uMbkk1o6auWBPGOadNLm1FLY3I1/d1ByZCK9KlpqZdWaVk23BPUVu2WpGJR8x444r2MDXbktSGiSS6a5IJLcetPjQnrX0tN3VmRLUsKgXB9KUzpCBlhz710wikiLXKlxq6R8gj8f/11lXmtMTgNx9aau9CktdDLu9QdwQpPNZczue9Nrl0KIFZ1bJNaFpe8gEmoTcmNO5vWV2DHjd+Zqrq1yCh6cd/yrRb3RMWmznZ5OSc1Xd89DQzVMiZiaaT3qXoxoYTxzTdwAovdhYidvemF8nmh267gM3ZPWk3gUOy2HYY0hJ60JljmtYpMexoWFqXkXvkiu48L6OZJFJBzkc/lXo4eF0cVeZ6rodrHplp57gDC/wCFeIfHfx0bxpLKKTIx84z+lddWXLBs56UVc+SMEDFOU4Oa8OTTiexF9x0mXHQ1GsPOaUHaOpCdiVYuOeTUbxnHIovqU2RmM56YzSt8oGOldEdbWMplO4kxmqjknmupR0M7jPvZpuwiqt3FLyHoueKljjO7OTxSdkNLqyV2wOaYrkHdzWVrI1g7anUeFPEDWsqq7HGR1r13w7rYkjX5s561orFTVzrrO6yBzWhFJu5rRszaL1neSW0m6N2UjuDirTzrON5JLHrmolDW6G3dWKkg5qM1VkRYToKXdQxB05JpM+tHmMOCKQUne+oxD1oJoQbhSZptCtcTOe9L9al3HYRjzSZ9aaBi5GM0hbNL1BeYu6lyM9aLDDPfNGaAbFzmjvmk9wFNANKw+gd+KM/jVDDr1ozipXYQZ7k0lPzHYXPFNJpISQmcU7zMe9U9gsIzk03BNJWDYOlN60k+ohQOKRiDTCwzijIpMYwn1pKFoLoNY0w8nmiwxjComGamW4XGOoqCT6UlcdyJh3qJunNS0Ve5CyZNMki2jJNZvcfMVpFzzVaTNRc1REwPeq8gwTWM2Mgc9fWq8i5Brnnoi0VZV4qpKvUmsXuUtjOuUyapTRbqa2Iloys0ZFOtH8uYeuapXY09Tt9FlDQrzk10dlJgVzve6IkjXtnI5rRhk5HNbwkkjnktTZ06Y8c1qPN8nJ7V6VOV46GElqc9ql5lj71g3Mu4152L1kaxWhBtB71HKgArlasitSo8gUnmqlxdLzisXKxdncz5ruq8l4DwDWEqjtoNREjkL96toTtqG9ddiRjtzzTWNZc90BGeTTlwOtRF6WZI2Xbg5rNuowcnHWq3egJFT7HuOcU9bPbztrrpyshp2ZHPCwGRVWUbR712QqLl8zVSK6AvIK0IrbK5qK9XXQ2uSpbYbJqwqheozWEqraGmXbJMjkVoR24YdBXHVabNegptFBzilERFKN+pk2SLEamjgJ7USSexLkWEg21ZjhJArK9/dJvfUsRW5z0q0lscVK0E5EojI7VIoK0XWw0ydMGp0qW7O5aLMWeuatRscZ3VaqNJFLRCiU5xmnBGY5FbOTdri2LcAcdatxEjmt6c2jOTXQu28pHrVoPuFdMZ9DnmQzjK8VSYMrVTTumiU+jJ0+7UcgzzW92Q9ys7YNRlu+a1jLQGiJ5Md+KoX+sJAhG4Djrmkp2V2M4TxL4wWNikbFm9jXJy6zd3RyWIzXHVnHmv0LdkQO7sMtkmkSInqOtebKV5cyKXcmjgGRVqNMVhJt6FxLEY6Vdt+DXHWlfQvpc0bYEnPetCNcgZry6rMnuOPHAqRVxWNmxEqE5qdPWhwaXoMbKc96iNJq/vFajkGTmtPTxlhmnDlTTYmupsB8R9ahQ5kJNejGya1M/MvwNgVchlx3row0+Wab6GcmadoQQM1c3rGMlgB9a+zw+sVIyKN5rCRA/N0rFu/ELOflOc/l/Ot73dky1HqUjeNJyxNRtLuOc1sm46Dv0I2fJ61E5GanfcCJlzzSLmNs1L90Vy5bagYhg5qK91AyrgkHNaJa3uNR7GbJLls+lMLZqW77GiGk+9NYnrSutwQx34qJ5Mc5zUxVmO5Xkmwai80+prRLS7KFEwI5prMSc5o5QfcVBuPtVu2gLkYFbQ3sQ2dNoGmNI68dwf5V6p4T0MBVdl6Ad8elevh42VzzazvKxH8SvGEPh3RnUMFZVIAB6mvlTxLq0up3Mk0jFmfk81jjZWjY3oxVrs8odCe/1psYO6vMvF6HdYsiAtStasozjArBTd7WCSWhHtI71G6k85rqglLUhz11I2O3JNVZ58V0RirkOWpSeUsSTTCe1b3Rm27kkcW7vUjW4JpSkNdwWEluTxUu0IvrWb12KuyJlJ96TZjjNS5dhrQkgbyHDAniu68G+JNrBHfvirU9Lmyd1ZnqmiaqJUXLHpxXSWtyGwc5+laboxbsy+jgjINTxSleAacfMQ9mBFMPPNFtAYnQUg4oaFdgc0Y9aSY2xM8UopN6gIeO9ITTsFxCxpOvWi40+wAAUpPFJ3ASkzzQwEIINB4pXuFwoBxV9CtwHNL15pbCYvPrSjnrSbAOaM9aG9LDuH40Z496VxLYN2TzS5zS21HZiEnFJmqTVgvoIT60u4elJ3AKKQIQ89aM+tJAITSZBoH1E3UhIqmJiAZprLSTFew3vzSE0wGlqYelKzC4xiKaSTSurajIpCeagcetJD2GHHOaiYg1LVx+ZGzYNRyNkc1lawyrLntxVdwcdazntoax2K8nHeq7+4rKT0KjsQSEHPtVdzn2rC1tGXqVpD9agkXIrKcehafUo3KdsVSkSpgnaxEu7IHi4Oag27XB96ae6J8zqfD8+QFPeurs+x7Vk5W/r/AIIT0NW3fAHWr8L9Oa0jLQwaNfTZDkH0q7eXOyE8130XaF2c8leWhzV/dbieeKy5Zstya8qrVu7myVhVmAHWql5fBAcHmsKlToty1qZNxfEknNV42a4bANc6lJuzHfTQivrSVBkZNZZMivgijVOzEpa2NKyiYgE1cYYXrWUpdCb6kDdaYSTXM2x37BmkZ9ozmqvYlEXzSHpTls2lP3TilfoKTtqi3FpQAyaSa2SMHAFb04tq0v6/AjmbMi+AXNYt1LtJGa6afY1TaG2eXcZrft4QUBxWtRrobxldEhgPXFPS3B61yT2sVGRZQeWKt28mCMnrWMOqZsnoWsA0nlZNOL6GcmTRwA/hVpIhQ3dtmLZKqVZhiBrNJJiuXIYgKtRxg1pZE3HGD0qtdKyc84rKpGzui4+ZBDM4fqa1LVt45ob5kbJltBiplBqHZpCvqSpGTVmGM46VpHbUUpFpFNSKcda6E09iGWYXz6VZV/et42RlIXIPemtEGGcV009dzN3I2AU1DMw65raK0JfYzru5WLOSOKw73xDHAT84AHfNZzqKBUUZF34xjCkeaB+P/wBeuO8Q+LZJWKRSMd3oa5/aKehdmmc9Ast5JvkYsT61opbgLXLVqR+EnVsQw5OaXyjmuXmVzSxIiYPNShTUVHfRFj4gS3WtC3HvXHWehUnpZGnZjNaUcfHSvKrPUyYGM5p+NoqVF2uhXuKrYOTUyygiri1qmUtRkhqEsQaxavsOJNE3IrXsFyAe9XSjHmu+gPYvsDtpsaknJrtULR5jK5ZD7BmrVrJu6mnCqTY0IrzyFznB/wA+9UdT1/A2q5z35/8Ar19fgK96dupKWuhgz38kzElzj61EsuK9SFuoyQS570okNaXE0Bkz1NNL1LethpaDSe5NML0NtqzFYY0mOc1C7FuhpqTQ7kbHjmmlqNBjS/HSoy/qanYF3I2Ynmo2fA5NWn0GV3Yk9ajJx70JdC7oFz35p6qSau3Yh7k8URY4rc0iwMsijrzXTSWpnUdkejeE9DLOgI9D9TxXdTSxaLpm9iAQmf0FevTVonmyd5HzR8WfGr61qrxJLmKI/meK83nnVifWvIxcnObsdqg+VHn+c596QISc1xXszti0XLU7Dk5qeaVWWt4QUnoRPczp5RHVVrjvnmumEVEytpcrTXBYVRlkZjWytclojIJHWnKp4rTRom/YuW8fc5qUgbc1jJ9Cle12RGYKaUtuGaHoadNRpznJ5o6nNRoLRjXz1zVixumtZA6kjntTTRcZLmPS/B3iXdtDPkngc16RpOoiVB61tBpq6FJG9bXAK1bjk707dSWTK3rS8nimITGTikZec5p83QlMbk+tHXvUsoKCaXUEITxxTehzTbABS4PWi66juJnFB60wDJoJ5qHuK4E+tJ+NCWgdQo+tNFBjI60uaL30C+guaM1LdgFzzikz70nuAhNAIIptAg70uSD1oaDUQn3JooaGthKXnFJiuJzRmjqO9wJpN1NIVxM9zTc0W7BcQmk701sDYoOPSkZqmwDPfNITTWomNIprUmDGNx1qM570mMYwPeomGKnqFyJxxUTUmupaInGahYEmsp2LIZF71Xf3rJ7DTK8nWq8hyelZyNUivIB9arSZBrC+upa2K8hPU1CxyOaynbUvzK06ZHNUnTk1MZPYl9iu6ZPNQvGPxpc1tyHpobXh9zvFdlatgDJqKm4T2NG3fPetCBs4qotWOdmtYPjAPem6rd7YyAfau1ytSZlbU5y6uM5Oazpbg9TXgVauuhrFEL3zYODVGednPU1jKpdaFaFOTcxxnrWvolhuwSK1opuSuCloal5pyNHyMmsKXSQHzgda0xUI25lv/XkZN21JUtvKHSo5VJPIrzqk0mJMiaI1C6kGs97FaXEVGPapo7F5O1atdAbRes9IJOWHFX/sCQLnAFarD67f19xhKTbsUL26SAHkVh3mr7icV0qmdEKempj3V00hJzWbOCSTmrp7jnoSWJIkGa6ay+aIHFTiZWV0OnsWdhNORMVgrNGi7DiMmpogQQaytZ2Nb2Lac9akVwvWrVuhmyxEQeanQ0vhWpmyWPk1ZhOKmTVwexbicVYRvWmmkyUWY+RzTZbcSDmioildEK2CZz3q3DDsFccrr4TRNvcsovHNTxqat+8rlIsRp3qxGlaU3K5Miwox2psinNdUYroZ3H2+d1XFz3rpirrUmTHFsGl38da2pOxmytdTCNSzGue1TxCtuD8wB+tOdT2aHFXOT1XxcuTiTJ9jXG6nrs13IQGO3tzXmVKzqKxuo8pls8rjhm/OhLEy8nJNKVay0M2aFtZ+WM4qcR881npJKRSVkOEXtSCGsZK8rDvoLsxTgh9aia5dS0TRIMircIrjqt7MGtTStAOOa1IyMDmvMq3uZvcdjuaa5qofC2R1GZGamh+akkpSsir2HSpgVW5pum1qNali2jJIroNLiBUbvw96KcddRS2L0yDGQKrrwa75pez0M4rUkZSRU1uCuM1y0Yy5mDLEp+TrWFqEe1s+lfU5XJp26GaKgBPenYx35r3ottlPsCk55NPDYGa1i7iEL+tMLknmpXxXKSGM9MLnNU2wY3qaQj3qLvdiaInznOelMZ+M0cy3ZViFpO9MaUHjNUlcqwhbj2qtLLhu5qoKzFYjBOdxoJJ5p77h1HIM81LGpJzVxV2HXU07C1MjqcV3HhjRjKy+vA/HivSoUzjrzseoaBp62Vukj4wB1/KvPPjH48W0tns4ZT5jr6/Su2q+SLZy01do+ctXu2lcuzZNY7Mzk+9eJdyPQ6aHGxgE9eanSPI6Vg4X3OmVkWFiITrVS5lKE4NdlGHLe5g5XbM6aQk561Xd85B4rVijYilyRwar7Wyc1a2EwUE1ZigLYOMVTelybWZaWPApGXjpWUn1L0IGT1oQHbzQ3dDWugFSD1pORWfNdhoNYkc0cgdaGrbB10NLR9Rksp1YMcCvU/CfibzYkUtkn1Nb02nqXJXid/p18JUB3A8etbEMm7HNaX1J0LSvmpVY9aRFhc9z1pre9DEkIfU0dBRuO9kNOc0AZFIfmIRg0YpiuJnNGTSATJpMnNKK1KDNGeeafUQn1oOTTt3GKBQT8vNT1DqA570v3eaY9wzmjOO9KweYDJNA9zQHMB+tBPFTdiYD680ucd6bYxpHekDGnfUfQXPNG41NtQDdSE012JbELdqaxzTVxiZo/GkxB0FJnNCAM470h96BjelJ1pAxpzTTn1p36AmNY8VGcjrUiGMeajbnmk0UiF6jepYyJ/rULHispIpeZBIT0qu/B5rOS7GkSGQdyarSHg1hNdikVXNV5D2rGRrYrv71Gyj1rCVy+hBIpPbiqk6EH0FLYmVirIOeKicU7aXZLVy/orFZwOnNdnZvlQc0pahNaGjAT17VdilqOaxi7bl23vNo69KpanelsjNXWrWpWIS1uY8kxb1/Oq0hLV4E9Ru1yAxnNMaI9TWlONrXJckQhMPk1uWE6RR5yBXZh5RjLUdiR9RDNjOaYwDnNY4itfYUrIYYye1QvAeprhced3IVkRvBUBtSzYrRU27WKL9ppnmY4rRj00R9q7o0Lq7X9fcZtkh2QKScCsfVdXVAQGH51u9FoVCLbucre37zuRu4qjIST61HQ6tCGTHeqsindmpgzKdrCxZVhxXRaXJuQAmir8OpNKxpHpmhetcSvsdBIEzUioRVNcsroaZMmcdaryuwfqalXTM9C9ZOWHJq+gorN30FcnjHNWUWspN9QepMoOanjYihT7iLEbtmrK/N3ocns2XYeqVIi81k12GTRoasRr61UEDZYjWp40NdKS3JuTquOtPEe7qK6YK9jJj1i2dKfux1rqirKzIvcRjULylc09EwZiazqPloRzkd6888QahLO7AEgH35rDEzdtjWnoczcB3JyagFuWPvXlupc0exPDZHP9Kvw2ezk1k53YicQ4FJ5PPSmm0tBtIUx8UwqKJu+orIYwI6mmg4NTzaGiJEbvmrEUm081y1EOyNK0fcRWrFkrXn1uvcykSBT3pkuawi7EJdRiqSelaFpbnuK0pxlzbA2LeRbBVRU9avlfO7ihqW7aPLCuh06PYo4wRzmlSi03MJMszqccCqpXBzXo8jcdDNSsPRxxmnm4Cj0qqGCm7MTZG97nvWfdy+aa+gwWGdN6BsV8ZpduTXrcrQw2+tIcmrYkJyOlNYHrTbKG49aa2M1pyk7sQj1phbmpZTGPg8Zqu4K/So5b7jiyFiSaTbnmi47q42RuPSqjsc1SdtBJCKTmne9XF3EkSouetXbSAyEDFb04XFJq1zqtA0lpXVcckjj8q9P8M6KqIrFQAvOT6cV69BNK6POqyvKxJ448VQ+HNLYhgmwHGD16Yr5d8V+KJNavpJpGyGbue1TipWsjWjHQ5m6lD5x1NQW8e5xxzmvOUUVqjiUIq1bPhgGrKmrzOyTuie7dY48hqxbicsT612MwjcqO5Peq8rY6mhLVDt2GrJkYzR5YJ681s421RnqieC0J5NWPL2cVk2tik7u4v3ep4pjyAA4qJa7FkYG45NKy4FKQPTUjIJ+tNGe9A76C7T3NIR70o9x37ApYcitrQNaksp1yx2g9K1ja5UNj1fwv4hW4iQCT689a7Wxvd6j5jWvUTNWGUEA5NWVfih+ZDuPVs96cTmmwE5ppyeRSW4gx60mcUu4JgD3pD0zmjUdtROvekIoW4mJyetA6mnbUaEII70hJ9aL3GB9aUNxTa0AATjNBIPepQB15pc56mpbY7aCZNGc1XQBc0c96m4gxznNGM9TSTKA8d6TJprXcnqGaQtQ9RMQEmkLHNHUYZNAOe9PzBgfXNGMjk0X0BiZNIDSWoCE5PWjNAdQ+pppNO2oCZ5pMmhoLiZppqNgGkYqNutNdx+Y1hxURFIERvnHNROKjoUiFhk1C/XrWbetikQS9M4qu471m9FqWiu59agk9e1YyRomVpeOcZqo/Fc8r9TVMhcAdajbmsZajQ0jI5qpcLk9azYnoU2GCc1GwrR6kNaFjTDiYV2dicoO1ZNg2aEZ96sRs1KUuhm9i7aQPOflBovtGkdSRnip9lKpC5m5WZiTWU0J+YGo/L9RzXmSpuGklqLcZIuBmqs7HFJvVMmxVeTn6UC7YcA01UcdikW7JWkbOeta8Ufy1Ckm3chvoSmPFV5BzRy2YkRGMnpTo4MHJFdNKC6lGnZgAZqa5ulRDzj3rvhFcthSicvrOs7CQG49K5e61Jrhzk/SsmmtLm9JWVyAHnJoZh1rnd7lsib1PNQOvNEXqZVNh6R8ZrT06QoRRJ3Jp6GyjblzT0BJrlekro6VYtxISOlSFeOKUiRRGTS/Zd5yalPS5m2WreAKMCrcceabvYRaiiqyiVg423GSrEakEeKbs9REidasx9am+mpomW40BFOCYNJasTJ0XPXmrEad66IRshNkyKc1Oi961hG2hLkSDI5p8cmSATXZSdmZNdSwo3DjpSPGR25rpmiE0iF8jrVK8c7Tg1g3Z3LWpy2tSMwIzkVx+oxlnOa5MTP3uU1h2MuWEZpqwZ6V5st7mjLEMBHNWAKzco8xS2FxSn3raKvqJoif3qNjxWd2xpKxBIxFRbmJ4rWMbq7RVkSpuUZIqVXIINYVkr6E3NKwOCCTW3atkD3rzKz1M5aouiPioZ1waicOW9kZpjYV3OK3LKHMWeK0oTbi4imtEVb5TuPFU0jJasE+r6lQNKxtSzAn+ddFaW+1QByMda9DCYV1E7u6InJCzAjr0qhPIF+tfQUMMkrNGS1K/mtTWZj1NejSopIqxGxJphGetdaglog2GkYPWl96beg7sYeeaQ09Wh20G7gB1phkyaqO2oDWakJ96q9w9BjN+NMJI5NRq0IaDk5pjn3pl9CBlIamk4BpKPUCtLJUOTmny9ATuKATUirn3rWKQPUsW8W4jNdDo+n+dIuOT/+qu2nF3MKr0PSPCmgfMjEd/8ACuwvLiLRbBnZ1XaMnn6V6sVZHnq7lc+bvix49k1zUHghlIhj4PP3jXl11dFm5bntmvIr1Oao2d0I2RWklb/69Ot5ysg571mn2E7s49U561IMipw927nRJNENzOwG01nSPgn1rsumRe5WkcnpUDMXPIprlWrC45EPBwTVm3gy3TmtXLQTbNFEwuPSmOm481yTmruw0rK5G461AymlFvcbFVWFK2c0pSTY1tYYyEGmkGqT5tR2DZgZpCv40Xuyb3Ym1sc0qKU5FWpFN2Z0XhjxBLZTBWfC8V6x4d18XSD5uRWqkviY5Lqjr7K8EiAgg5rRil39a0avqZ6lhWyKkA9aTExfrTT1pLQTEFIeaT3Gg5pCTTH1EPH1pM4HvSWwhNxB4pQaGMQnmkyQaI7BcDkikx60noNATikNMWwZxRuOMkUh3DORSg4pPYGB+tITkUDFBNKWOaGNCMc/SjPpR0Fe4Z9aafyoQCZoFML2FNIefpQhIKQk0h2Ck6daCbhijNANCZzSZpjGknNITSuAnfmkJpPUBrU0jnNHQdxjDnmo2UCod76BcikBIqFutFi1sQvkc1BJnPNZtK9yo7kT/d5qvJWLXUpbleTg+1VnNZtpo0jqyvJyMVVlAArkle9kaLexA6981G+aybRpYMZFQTITUPQTRSkQgmoWWnotSGySxBE4rrrBjsHNEmSzVgyetXraPzCPSo+LUxludJo9koAbjNbbWkbRYYDpXp0YKMLGE73Of1fS1OSAK5e6txExyOa8jMaXvcyHG9jNnYjNUJya8edTTQplZtxNTW1q0jjg1EAvY3LGxKAcVorHtGcV2wppK5ne4ySqzijlTZSYIuDzT3ZUXJrppwG0VJdVEOcGsjUvEWAfn/Cu2Efd3NFTuctqWpvcMTmqkEpLZNZ1FpdG0EWwxIpTkiuRvQb00GHkUioSahMymyzHDkDirUKFWrOTV7GSNW2yQAauxR1lpuza+hYjGBUqr7VM2r2EPCVKiVNlsibliOImrUMRNOfw2C5ajQ9KnWPFOMVK1yW7EoWnBTmlypFJ2JFTvTx8pzWMo3KTLUUoIqdW3GnGKTHcni9quQR5rshEykyfySKeqYPNbRpWuRdPUcRRHGc81pCOoN6F6JAFGaJXAzit27oztcozS1l3twSDzWEmluapWOc1Js5ya5q9TcxrzsR71zaJnSRHPSkjgJPSuNOzRsyysPFIV21E4uLuhRYig0jEgVurJWYtNyFiT3pjk4pSSUSrELKSc0+CMFulCmuUEtdDWs9GfUCFU8nip9S8I3mmwrKy70P8QrhlKS95rQ05LrQpWwZXw2a3LI5xXPX20Oee2hpoPlywqpcud3tWMpu1jFC2K7puRkV0tupEIyO1VR2dgm9rlC7+Z+BnNMgtCzdKKq+GKQk0jUs4jGelbMBGzHevcy2HKrEvUgvCPX61lyruavo4LXQSQ3aAKawwK6OWwDGPHNRvijZhuMzk5pGcetW99CkiJ5fWoXnHrTTugI2uO1Is3rU63C2g/wAwYppb3qmwGM5ppY96LhuhC+BTGbvTiOwwnuTUTA4ND2AqyKc56U2lG9wSBQTVq3gLYNdK8gasbWk6cZJVyuc4Nd94Z0DcykqMDH9K9PDwuzirVGeiWFtFp1oJGHIA4/KvJPjB8QfKR9PtpPnPDNnp04rqr1OSDZnSjdngup3ReQljkk81jXLMWzXgr3ndnU9BgYt9KVDtOT1qrW2BHLK3fNSdK0w+kjao9bFK/kX1rLeXnriumKuZO97IZyx68Uqwbnzmr0E0y7BAp5arHlAc9KmSstQV76jgaazjPNc3LzO5sRNzzURG481XwoSV9CRcY5NDDHvWVujQ1oMcZ70gAWtdNkS7tgRTTjFNLsSlrYM54zTeneqdhtAshRsqcGuq8L+J3tpUR3OB700l1KS6Hq3h/XVuI12t6V1Nrd+Yo5roVhWsaEcg455qdZAal3J5WPzk0vFDE0BA60w0dQAmm5BoBXDIooY7aje+KO9Nie4ZxSHJqUCA5zQR60PYY080Dmn0Gwbik696mweYnTrS96Lhe4ZoP1oewBnHek3U+gxc8UgPNK4kKTmkpg9wzzSck0IGG6jPFCQbADmg8fSi4ATmkzSYWEHXrSk0bgN4oNO4MaeKaTnmkl1DoBFNY89aW7Aac96Q8U2IaQeuKiKnPWkmkVshjjtULjFZ3KIXHNQP1zWb1ZVyCQ471XlP0rK2hcdStIwPWq8p44rCd9maJMqyE55qCUg81jJa3RrEhJ9TTDisJRsyrCjpUbrxzWU42HboUp05NVivWqtdGbQ61TEw+tdXp/CDmlKSTsJmrb5Irb0m38xgSMiqgm3ZGMu50ttCUUYBqcNIa9ByaXumat1EuLUvGWb0rj9biCydOTXn49c1LUhPUw5YM5qnLanrivmZX3a/r7i+g2GxLtyK2LLTwuCRW1GD3IZorEqCmyNiu9pWuTYqyNuao2GaVOGpRFLMIhljWTqOsBBjOa6qcuxpCPMc5e600jHBxWXPdPL1aunm0sb6IrH5upoRgjdaSvytCV7luOUEfWpN2OM1wSVmayQ5F3NViKAE8ms5NnNPyLUUWDVhI8HNYyjdtk7Fq3ByDWjGeOaS+GwXJ0yanQVi7FEqLmp448kU3JJWRLLcMdWY46mKctR9CxGlS7cCuiCZDF6UBvWlLRBcmRuOtKzZ5qWlZIpDkJzU8bnPNTGPvDuWYnJI5rTsyxAB6Cu+jfYmVi8I2Izini1fqa21MtFsHkPnjJ96ekLDmrtYVyQkrUEpPXNROT6FIoXLHFZN6/JIzmuOc29TRGJfc5rDuRkniuGpP3rm0UU2iyelOWDFc1na7LvdCsMCoWYc07uWjBEZamMxzWsrqNh2GFvemHDVnJ6WGhpFKp2moGmdL4P1eG0v0N0CY2YZPp716Nf39jeWDW8K70IyGzkD3qZ1017Ox0U1pc8sukSK7dEOQpx71pad16/SuKqkjiqaOxrf8ssnrVGYfNXFKfM7xMUi3pUWZskdK3wcRfhW1OSUGRJalVLffISRWhBZqFzXThqUpWk+hMiVIDuGAeatqjIK9XDTcJaCZTupGbrVCWTB969+hVUloUn0IjNg0xpSea61NjaI2k461G0uOtF2wSIjcVG859aalcdiB5u5NQtJmnzMdr6hupd3rUOTuHKw8w0vmVfMriE8z1NIZQDyc1UU3oO2o1mpjSE007Ow2rjC/NBcU3IVugxgGOaYYsnrV07Bckji5rU06yMjrwetdVGLvcynJ7nbeG9EMjKwHcf0r0vQ9KWziV3G3bzz+FezSjZXPPm7s5j4o+PIdBsHSOQeay/JzjPAr5h17XZ9SvJJ5XLMzEkk1xY53tBm9BWV2YhkMr5Y5oeIuM5rgVkbkTQ7BUJyM1SkloLY5qMAn5qkl2qud1VBNbFzb6GLey/NVIjcTXVBPczTdyWBCatwxEMM81TSGm2tS6sYA6c0pUYrCTYdLkLvtNQPLuNKCLTbH4JXPamFcHNK+uo1dC5wM0hbdUruNPW4BeKTOBQHmNbNN25BNVzdguAQmmkYp3bdiJSGk0+F2jYMODVRui433Oz8KeJmgdEdvYc16loetJcInzjkdjW8Wk9AlodLa3Kvgg5q9FLkU2nezI3J1Ykc08HihsVhQ3amnnmk0FhpPNNJ70MBRikoHqDUnQijoJCYyc0nTrT6WGhf50haoauPoA5o6U2IQjNJkAU07ofQQnPek6VPkJC0hqnsNB25o6danUBNxFKWoa1AQn3pfxp3AQ0Ci4xc0nWkhW7hnFGaADNJ3zT8wFI70jGlcBAKOfWgN2Nwc5pKbeguoh+tNI45qb2G3YQmmE+lCEJknk1GRyaRSGOT2NQtnvmosrFETjJqBwamyQ7leQHNVpRzWT02LiVpExVdwK5pe8bJkEgP51XkUHisW7MuJCy4Jpm31FYylqUk0KEwaQpWUrldCvLEeTiqrw888VKdtjOQW6YkGBXSWGBGCaJRtr1IWxq2YywzXXeH7YyMvGa3o3ctDOpsdtp+jecgJGc1bl0RYV3nt7V6tOkmrnG5mFrEy26sM4AFcVqX7+QnjFeRmMrr2ZcdrmbJbj0qI2wJ6V4UqN91/X3FjorYKelXEIQVrRjyqyDdAzk1Cxyau62EyMjFRXEyxrkkVcdhLU57VdVCg81zN5etOxOa6Iq2x0QVlcos2W561EzgVopalxi2xhcdqiLYPXmqUmjop023qSRT7W61fifeB0yawqK5pVp2LsEW73q/BCMc1xzTvY82V72J0hCipNlYyulci/ckjBFXofmxmqppWsykWo1zViNO9ZO17Iq5YjjyKsxxVEttCS3DHxVmOPHetacHfQGyZVxzTiDW6fYhjG4pAMmpkF7E0aZqRYnc8Ck6bdmF+5dh0ySQAgcVMNNlU8rW6oSVmxcxo6dpTSkblxk45ro7DQvlGF4+telhaKtexnKb6mlHoo+tStpSqBxXS6S6mXN0K81kg7VXe1ABPOT2rlqIuLZA9tjtzVaW2Hf8q5JrsaRbuZd9HhSaxLsZJ/WuKcejOhPQyLpM5rLntcseK45RX9f8MaRdiMWfc1FLFtBNaWtHVDbKUxwarFqwmlfQpXsMZiTmm/eqZzb2GIR3oC5PNZ36CuKyZ6UwLzzStoNeZet2AAwRirq6vdJH5UczKD1wT/jXPKKbu3qWpNEIjJbcTkmtOw4IzmsJS6GM0a6cx8mq7R7m7VwzfLJ2MEtdTU0yDaOnJrRl+ROa6KK927RL3sLbAFxWgi8V6+GjHlsiZXJoQMg8datlQ6cV1293lRm+5m3cAGeayLmMqcnPPWtsLNxqJM0W1yqaTPfNe5GWg0NOOpqCUnNV1H5kLZqJwe9KI7kLHJpCeat7ajuBOKaSetDikwTbEz3NGeetOL94d9BGPPNNLnNNSYLUNxPemM3vVc3cZGX96TfTSbJWo9G5zUyJvIreC10Jd0XbW03uBiuu8PaL5zqdvcD+VelQpvQ5as9D0jw3oghQFhgAZz+VSeL/ABPBoOnPI0gXaMdevAxXpbI44rU+ZPHPiubX72SSRjtz8oz0FcPdEZzmvIrSc5s7owtoNtITK/rmrFyBAdorKpEOaz1Ku7eTzSGDIJNQ0E2cCmp7V5qOXUC44Ndqit0au1im0jSHLd6kSIHtVN2dkS1bY0bSzSRM1at4UR8MKh82xMVJuxLdRrGMqapmTnrWUZMtLuRTgHnNRRx5bNJN2Bp2J/uion+bmobuUhMYBpgPrVxRSj1RJ0Xmmn5h1pLdk3GMCB1pFznmrTRNrj2HHJqMjJpp2V0AwpzTwPlprUew+CYwuGBrtfCvigxOkbufSrUnfcrWUfQ9O0XW0mVPnGe9dJbXKuuc1u9rmaTLsUuRk1Or5pMHoOFLkUmhasaRk5pCM/SlYYh4pM56mmCE96CeaLFCd6OCc0uomxD60UaAgo981PoO1hOp5pODRrshARxwaQgjrTuCExmkPXrRcYmeaWhgA4ozSBBRQgDFGaaeoNBmg9Kl6MLhjvSHI5oe4gzSc9qaWhQZyaU80MQhHFJQDF5phpLsIQ801uTzTGhpFNIpPYewh55Jpj/WpuCGEZ61ExxSAhb2qB+tQ9y0QOOtVpPWsJFrYryc1WkAJrnlo7myK8gyajIzmsJlELL+VIIxn1rNtPYu4eX3NIVrG61AjdMrVV4+TRFIhsbCmJM1t2fQVNWTC1zZslwQa7vwqoJXNa4OV5WOas7I9I0mNFhB4x2pms3KQwEZFe9F2Vzh3keba7e+dMyg8VhTHJzXzeMleozoRWdRnNMPHWuKabiWgHSlB9azStqJg/FQsTmpnvoJshuLgRrnIrA1XVAoOWrWle+hVKDk9DmLu7MzHniqEsgBznFdSPQhR1IGfNRFiO9WjrVJIa3seaaMA85zVI0irBxnPetCwjyQeTUVJNIirJqJu2sXFW44wDXGpczfc8eejZOq5qRU9e9RJWMWSJH3NWYR2qZK+qHfQuQrVyKIVjUVncu5ZjiqzFHQo9QuWkjFWEStoktj6a1aaIhojPLZNPjTNZpXkN9i1DHmtXTrMSMDivTw9LnkZzlZHVabpke0d/61eOjI3T/P616sqCaWhgqlixaaYkbgFehrXhiVFGP/ANdVThyITk2x5IHQ1DK4qZx6jgtSnJhuah8rLE5rinqzURoRjkcVRu4gq8dqxcEXGWph6irbTx1rnrsnJrzq8WlrudEdSlJFu61XkgHU1xOF2my9yvLGAKyrxsE1c21Epb6mZM53VXdjmuSpJPY1sIOnvS9ulYOTvcSDApAeetHS5Fx6g0jqQDTg1exSepCXZehNSwSnO7k0Pa5cpGlbyoBuPJrQtJMsDXDWXvGUjXiYFBmlRN8lcNXV3W5i2bunWuUDBTVya04Ga9XD0XKm0zFy10KsClZeeK04xuTNVgLpNSKnvdD0+U5qQz7BXq7K5nfUrtMJjwap3luSpNTRtJplWszJlXaxzUbPXvU2rFpkZfmmNz1qxjGGTUbR5zVryEQSR4qEmk+5T7DS2aTqaqOrvcLhz603PPJo0uLqJnIJJ5ppbNa36DuNDe9ITU/EVcY5wKaDzW0I6InzJUGcVoWduXYYrphHVIictDqNE0dpHRtuQcdPwr0jw3oRQRlxyO/5V7FCGl2cFSRuarqcGi2T7mVQi5JJ69M96+bfih8RX12/khilYQRnAI6Goxlbkhpuwpx5pHm9xdmVick+pNUpZAXxXm3bVzq1Wgsdy0J+Xg0PI0h3MaJS90LK2oKyjvTt5PShJ7sl7HlIORUsalu9drdnZFpX3HiMk/Sp4Y+MVK3FqXYJTAMg1N9r3HOeaKzTWhSXUbJIzryagbHSuWDexUdRrAmlQ47VTdxqOopywzTMGo62Ew2E9aTGOKvm6IOYUZ7nimlcn2FTezuP0GE44pV5PNU1oK1xzciosHdyaadtAaHbc9etI1EXqS9xucVNBO8Lh1JBFNN3G9Edr4V8UmN1V2PvzXp2ja2s6qN2c+9a023oOWh0Vvc7sHNXI5M960uQiZZOKep4poBTSZNHqJ6DTmmn9aEh30G9BS5yKBsAeaQnk1IJBnijNIQ0nNAyKpFeQucH1pMGpT1EHSk5JyadgDrSEHNIPIQik6Ul2KE75zS9eaNSfQC1LmgAJJFJmhDT1DjFBOB1oAA2DzQTQ9wEPFFNh1E5zS5zSYMQGjOaEDAj3pCKYuomKQrikwGsuaaRxU7h1GFc03ZkU7juMZfWonBqGy0QyA96hYVLasBBIMiqsq7eSa5ps0iVpBnmoGGKwls0aldwTURrBmiI2GTzS7Mc55rJ6IpiYPfmgrnr0rCSsxWGlMiqdwuDwDTgtbBYbEhLD3re023yuetKq9DNuyNSH92wrqvDl95Lrk4qMJU5ZmVRXR6FYa2kVsMkdPrWH4i8RK6kBxXuVavJTujjjDmkcZcXPmOWyeT3qq8lfPTbk7yNrdEQuxqInnk1HNcNh4pCe9ZzaSEAbcaSQDaSKyd5Ce5zmuX/AJBPzVyd7fmVjk100bpWPSwtLS5myTk8ZNQyMSRzmt0j04wSG5x1NMLjk1aQ2ho+Y0uMtzVDsSwReZJW3Y25GCK5sRLSxy4mRqxRYAzUygE1zw8jyKjuyaMHrU6r60Ti2zG47HcmpIsk8UXtow5tDRthuA9avxR8VhJc0ilsWo1xU6DmtIobJ4lJq0qcZrVK60JbFINMZPWk1Ylsj2c81NEoNTTWoS1WhKJNrDFbOlMchs162ElZmUlodFa6iYtuTjHvV1dcQdWH516Kr23J5LosR65F13D8xVlNZjwAzc9j/k0/aJvQFTaJDqiddy/nUZvlY53j86idQqw37QrHrSGdV6MKxlFbj1I5bsYzkGs+5utzdRWEtVoUkUroh0JJrmr0KZDg8VzYqn7lzaD0KjgdiKrzkAdK81Q6mhnXU2M1jXUmc1z1dTWJnTtk1XL1xRimzXoAanrkjqaJKyuS+4m0k1IkXGazk7GDZKkPNPMXrWaepSlqQTRYqNEI68VpfQbfUuwLzxWlaZXFc03Z6hubNsNy81pafYGaQHmsPZuVSJhUdjqrLTSsQJHbjmlmhwOa+hjRcYanOtWZtxEQ+R27VZtDuXFcFNOFVp9TV7WLBQ9ccVFPGx5Ga9Z07xI06EEMDb6tTQDyjnuBWeDT57sbkc3foY3I9KpM1e3bV3NFqMJzR1rWK0uwADHWgj0FOO4yGRcg1VkQ5pTWgXImBFIePrSpp2uOwxzUbNzmtopMb2G78HNG84+tU046sBAafyBU8ruLcidiaWNc8V0046D0sXrS2LEcZ5rptF0hpXU7OpFd9CF3c5akrHonhvw8E2bhgHv144rq55odJtc5+6AD2r1Y3SscV7u54N8XPiU16z6daS/KOJGz16V4vqE7yseSfr3rysVNSnZ7I64xVipvKjGeTUJJBzk1gpeRTY4Hpmpc5FDiL1IWbaafG5JGaevUNLnlqdc9alBOM5rsWkitncsQoW5q9bxHFKTTbsS2+o54sE5NNjjOaylJ2sap6Dn46Gm4yfmqL9eo4SdrD8cZoZQFzUPQp9xARikboc0N3ZIxmyMZNIiFjgE81cVZWFZkrQFFyaibOCKlu44kWMdaVUJ6Vo2mOQ/aRUZGTiqdrkK6ADHSlxnrSvuL1EKDvSt0AApLXVjtcfBM1u4YHBFdp4U8UbGCSOeOnNbQva5W56bousfaEBZq6G1uN4zkVtdNaE6JF2OTJqVWyetKwEm6nZ4pbi3EHI5ppGanYBp4pCcU+gBnnNHWl5jQmPekpbi6hz1pp4600P0FBBFGfek9BBmkzzTGGcDige9LbUOghI9aQnJzSVwt1Y1vU0A+tOwxcg0Z9anqAZxSZzVAg+tGe9JgA5NKck0aAIcetFGoCdTkml4FHkLcM0g5NGwwJ/OkA460J6BcX3pp5pCGmkPT3pDY0gDrTSOaQDGXNRutJtMaIpFqBxjNQ0NbkEgqrMuetYTWppHQqyLjvVWTrWEzZakMmcc1F9a531LGYBNIRg1i5dCru4Y59KUjHfNSxvUY3vVeblsdandkyJ7O0MjAkVvWsPlrUVZJKzMmyf61atLtoGBB71zL4lYnc1f+EgkEe0Ng4qjJfvMxZ2JzW9WrKWkmRZIjMuaYWJPJrFy5tiLaiEE00is2xNjTJtqIyk96460xEkZJxUsq4hJrbDLnElqcD4qlIlxk1zUhJz711x3Z9Dh4pQTGEcVG/B4rVG/UbwR1puBnBPFWgE74pcAGmBZsx8/8637Rgqg8CuWsrvU4cTuXFfuTUyEdTRGKWqPKqMsxketShxmlPfmOdscORU0SciudfE7ibNK0TkVpxr8uafK2y7ki1IhJPNJJq9yky5AucdavRxHFdFJaCY2VcVCaXLYlsbipEFRBa6iHiM7s1p2FxHAvzmuyhU5G7kL3nYkutXjRD84z9axbjxFJGTtk3D60VsV06nTTp23IU8WzqcFm/A//AF6nj8aunDOfbn/69c315pao09n2LCeOD/z0/wA/nVu38aqer/rXQsW3oQ6ehfg8YRueZB+J/wDr1aHiiN8fvAPX5ularERav0EqdhG16P8Avj8//r1DJrMZyd/P1H+NU68eouRlG61r5SA+fxrDutTLOTmuSviFJWWxtGJCmpA8E055w4zmuVa9RtWKNzgg81h6jcLDnJrKtT7DgmzHbUEd8BgaeG3Abe9crppK8jZ6EqL61KormqNWsiGSLHnmpo48nrXPKTMrFmODPNEkPU1EJczsC0IGhqNocGtZrl1Gya34PWtK3TODXNU7gtjXs8kjNdZoECuVJHp/SuvCU3UqJs5q7sjrooo44RuC9KyrzDPxX0E4x5XYwi3a5RmhVh05ogj2mvKnSXtOaxqpXLQUkZprJ616sEnElq2w6ONQegpbgDZ9KiguWrZA0zltV4c+361mNzzXqpa6m0dhuKB1qlK25dtBeKQnimu5PUjY8c1DIOpq3G4ys/BqNj3qI9guQseetNJz3q472KYxs0nJrb4lqJkqLkZzQ7YFOK1sIRFz25q3bW+8g45NdFON9CJHRaJpXnSJxnnJr0Tw3oJAVtmOB/SvWoQtY46009jsFEWmWu5tqsFGOa8j+K3xKECyWVrKd7D5ip5HSt69RU4NkU4tnhl/O1xIzFiSxySetZFyME4rxndvU6Y+7uVhGxbJpJBjvRcb30EyFHJ5NPUgjrT6XJ8yNoyW5pRkcCs5N9Rpnl25l44+tSx5xXotJK5d0kWoX29TV+GZR1bNZyVloNSUug52DnOetPVSBWM27aj2E57UhTjJ60krAnqIfu0x2IXrWblctWuhEOTzSsaqUdRytcZj5qt2mwcntQ430Ba6D7uUHgdKpOCTkDinyJWSJ23BYCetTJBsGanqS3pqRSAk+1RlTmtvUashNpzxQPl5PWi3YV9QBycmgnBppdBJkbE9qkgmaFw3pzTi7DjI7Pwv4saKRVkb616domurcopDDmto6bg9Nzo7e5VlHJ5q4knHWrZN9SWNieTUmSec1D3G7XuKCaRuDQCQ0j3pGpCe409aXJpjvoITk0HNC00C4lJ1oC4c0mc0lqCeoZBGaKHpowAmkBIpDFB4zRmlYQ0jnrQRii4XEpaQCHrzSY71Qxc0hzRbXUVxR9aQk0tB3Ac0tDYMQ9c0D1o16jsICc8UpzQ7CE5NJz3pX0sAUHnvRuAlIeaBsQ/Wmn1qUJEbetMalYoiaoXXqcVm0VfqVpFPWq0gPTFZyKWupVlSqrrg1zTNUyCSoG6nNc7ukaITvSY96x5UWgA9TTtpxnis2txkUgxyaZDAZnzihOy5jOT1ubVlabADiroUgVyzfNqzG44LxzSqKjZibHc0ZINDuyB4yRTlXHWq0imQ2KRg1G9c85LlJKzsSaRFJNcE23K47F23h+YE1NdxEW7H24r18LTSVxQequeZ+KGJuWB7da59iWqo2u7H09D4EIw6UxzitEaNEZY5pOTzVhJABupwjxQ2TzFiBX3jHArYtSwA5zXNXtbQ8/GSSRY8xh7VJDOzHrSpzTVjx5SLkbnsakjdietTVnpYxdi7ErMBxzVyKLaBnrSjHmuJsvWQJetRVwuau3u3CL1sNJwamh9zXPzNnQjRtwMA1eRxitqbRDZXnNQge9ROpqO48ISacoKnpmrjZ69jFysXIkDrmqeob0BwSM1vVj7l4mdKdpamLPPJyCxP41Ufc2TmvKc7rlPUUtCF1PrUTZ9aIpq7bEQszrk7iaRbuRDkE/nWrm5NNlXuPj1SZD94gemalTXp485ersmtHZf15jViT/hJ5B1kPv8A5zUi+KGJ5k/KiXND4f6/EOUl/tsyrnf196ie88wfeqaj03/r7wSaGxytnOan+2EL1qFVsr3B9yGS9O055rm9aaabdjoelTKveSFFpO5hW1ndGfJJxmuitYWSMbjWmJrRcUPm5idVOc1Iox3rypNPVBJk0a561LH1HFc9S71MtS/Evy0Ome1YUWufUGyKSPHaoJE46V2V+iQ0xsCHdmtmyTIGa5XdtJIGaNunzjFdhoLYVD3GP6V6OEjyTbOevqjcmumZBz1AqrLluT3r1E+ZmViF171Gow1KpT1HFlpOeKSQV0UU7Axi9alkA2c04QaqCkzl9XTBznJNZRFd8rp3N4PQiPApMmklfVlti5xQcdc1qkJ7Ebnj1qJzxTabB7FSTOc1C7ZOPSlfsPRkZ96YxJ6VVNdx3E5qVEwMmtkJvoKxC0xAWatFDmQi1b25kPArotH0dnZeM55rtoU+pjUkkj0Hw5oGNhK8HH9K7W2hi0+2DE4JGOOPSvYpqyPPldvQ8z+KPxJj0uN7aCQNO3HHavAda1Sa7laaV8sT19a4MbUu+VdDqpKy1MU3zsxyagklJbNcitbQ0drjQ+Op5NRyqxPWk1pcSlZ2IsknHNWYVA+lFt3fQcrPYe43ZxTGjKDkc1LXNoO/Q8qz6U6N8Hmu7cltEvmDPWp4396G2lccG1oyxHPg8mrSOZOprFpbmtx4BBpxIA561hUug0b0IJG+tMOT1pK1jW2goUg8daMHueapyVzO9kG2lXIHtQ2NsDknk0qjHWnzXQPuTLwKkCs/Aqla9yOZdRjwFevWq8ue1Q3rdivoRBiOtB4XJrQNhvOKTaTz1q72HcRgVpACanpcLoljmaFgVOMV1nhjxY8EoSVyMd+1VCSe4PVHp2g+II7qNRvzn3rp7a6VwCGFdCfQjXsXkkLDrUivxSGh6NTs+9GwMTHfNNbOeaVrBfuIRg0g4PNS32DcCRmgHNGohp60U3sO/UMn1pppIa7gKM03rqxNgKTNIaF5pKHawXDPOaTPNLSwwJIpNxIoVtxBk0lF9Rhz3oFNsNhelAOeTUMNwooC4h60Emm5DAHFFGyEGaQ0uoIQmgHApagBOKbn1pjQhpCPegTGnkUw5B45pW1HcY44qCTms7LUaK7pjNVpVrJtFplSYdc1Ul75rmlsbRKsvFQtls1zy7s16DVGO9L3rJvUodtJpSOKyl5DQ0QlyK0LOy2nJGBUOVkYzdjQVQBgCg5zXO371mZ3JAOKXbxQt9SWxcU4Lzk1La3IY9Vp23vRJaCEZajZM1x1PIgiMRJ6U5IwDyKwjHmlYXoXYE6cVJe827V7VBWiNWTSPMPFa/viehrnWUmoSSbsfU4d/u1cawOKjcEitEatiBD9aCMCnclu4Be/SnovQ96GzOTLdou6taCH5BXPXasePjpWdh7RE96YgKvk1z053djyeboaNuM96tQw/NmqnvYhbmlEAgqUSDpW0NIoTNHTU3OD1rUddq1vOKVMmEveK+CTU8CHrXna3ud+li5EStWUYnmtk3Yzk0PKFutKID6UOCerMpTtsSRIFbnFOljDDOa0gl7J23OeTvIW3B6Zpt9HuQ1pz3hqhxdpXOeuYwrkVWZK8mXxHpxloQsvNQuvNHQp7kLLmoXTnpVRkluMidST1qFl5rS7cbICJ0IqEg561SbtqUpMek7qepNWo71gQCeacnzWuVfUuxXO8U8sTXmVqnLLlQN2Y1uevSoJIUcc81zXlcRElqqnOP0qUpxXRvuJ6bCBeadjvWU+yEyWI+tTDgiptdaiLkMy4AzU/DVg1GMvd6iIpYz1quwycV16Sg7iTHQQncDWxapgCuWCfQV0X7dMuK6rRB8n3vT+lethktTGoa5XK9ajJzXVRS5tTJ7XI5U71Hjmulxb0QJliEZ5qR13DmumjBBJkRTB60krAIcmtVD3rkt9jmtVYNIQMEVlSKee1aTXc6YbEbLxTDgVnBWZQZ700tWqYNkbvgdaryyH1qo6bjW5WduOtREnqTUKaGN96FGetapt6ibJEShmwKuN7hYYFLNVq3ti54GTXXDUi9joNH0hpWXjuM133h3w8MoTxgjHH0r1cPDsclWaO3tbaPTrUOWHC/4e9cN8RfiFFpFs0aSBpiMAA9K6pNRjc543PnnxDq02oXclzNJuZj69q5q8naXqeTXly31OtPQrqpU8nIpz8rnFc8nqU+4yMYbLVKwyueBT5rxsTbUi8vnOKUDBJ6UlHoinYlSQKOOtEsvy1HK7jWh5KTikB564r0U9CUl0HB+/UVIs+PrSb1sOW5KkvQ96uQXm3qc1Fkykky4tyjqOxoDKScc1jNO7LvoIy985pp4+tZRTLTuLuOMmmkkjrVJIloac1JCpJwaLPqN7FhLfcfanvbjHyg5ojFsylPuV2Uo3JqZJ1A681pa2w0riS3QZSKrklu9RNK12U7WIWAJpu4ZxmrTbIv0HAfpRyOelCV9QuhhOTQAKp6IbEYUqSsjccYoWmqFeyOm8O+KJLSRVLnPrXp+geJEuUQFwT9a3py+8pr7SOstL1XUc1dSXI61bJv1JVf0qQNmhrTUL2Y6kzmp16juI1MzStqJCn6U0n3oQW6iZz70E5psBRx1puTUdRphnvSFqqwADmkNIVxaM0txjc5oz60MrZAST1pAaXkheQufek70AGaMik7j6ATmkpoQuaBzSTBic+tLz60nuITHvRzVXGJQTRcBpNKT3ofcbEyT1pD1oYICaQiodwTE70hGKTYiNlzUTrjtU2Giu65NVpV5rKcSrlWVQASapTLXPNaXNYsqSDPSq8nHeuSdzeIKuadtrKTKT6DlBAzT0jLnJFQ7CckXbe2AwcVcVcVhPaxjLUft70Y5rCo1cm/QVaeMil5g9h6DinqpNJa3IbRIsfrTvLz2oV2rEtgITSGDca56kHfQybA22ad5IWnRopPUl3uPjAFQ6g+2BjmvTjZRNoL3kebeJCXmIJyRWCy8muaLV20j6aivcRGTtqN2wMmtkatDA3bNNLAYq0hWFUg9qcGINO2tjJ6PU0tPQMRxW1DHwK4K54ONleVmSsmBnFVHGJM1jSXvnmrcv2abhzWgg2it5tqQuo/zCcCpokLMDW1K9iJs2tOPljmrsj/LWtSV6dkTDcZH85rQt4uOa5FTZ1SnZE+zHapIOvNVBpSszNzui2sYxmphHla1VmzCUrkEqkNxT40JGetTFtPlQ01a41nEeeap3l6AhGauTUYuxcVqYssm+Q+9IUyOa8xJzep3wasRPHUEiHvTlHsaJkLpUTp3qGtRohdMiomTitlotAZEy1E0fBJq+a61Q07EZXFIRtOaGuwXZYt7jHGavC5ULkmuWth25D1sMkvFA61XN4M8GnPBrdAtNx63KtjJqZXDDrS9i4rlG3pccBk02RtgOa5a1Hl2IbK5vfLah9S461tRo32/r8RaFnTrku+Sa2oG3muHEU2qnujJ5IQUyaouuHxXR7P3bIhO7LNunStO3TgdaxlFg2XYRgg1u6RcbCOeP/wBVd2H5kZT1Rs+cXGM05ea66TfNdmb7DiuRzULLg5rvtzaohMkR9vSpNxNdFLRXGxjnNU72Qohra13clbnOXU26Q+tVW5PNE02dS0InqJjioaKGlsD3qJm96uOiuHmyGSTHvVeSQmieq0HbqQs2aYTUQWo1uKBmpAMDmuhNLQTYjSY6UgUtya2ilcVupZtrYuwGK6HSNFMhU44JrvoQ1MKk7HfeHPDxG0kHt0rtLW1i06EOx5x+NerCNlZHC5XZw3xI+I9v4csn/eBpcYRAfpXzjrPjC51m/e4mkJLnu3b0xWNV3momsI3RnXN4JVIU1R/iOetctUtaaMfFEJOtLNDtXgVwyWtiutmU5DhuaPPAGDWt9DRIfGxfoKe6ED61CnqJtWsRA7KazE8mm33F01PLc5o2561121uiY3TF4HFCnFPe9ym2hwY+tPWSiK0uyr6E8cpUe9SLckHk1W+44vqWo7oMOamQiQ9+awqx7GvS6HGPGaaI6wu0S3d3JI4S3XvVqKzw3rWkU72YpO2qLDKIx9KqyTEtyataaEJJkMuSuTzUI4qZblx10HeWzU0pgc1ne+gS0IgmSck0hixyKrma0HNDiuB1qMn1rSOxCExUix8ZpSd0EhjoQeTgUGLjrT5rEqWgikxng1veH/EMtpMoZztzVRbvrsXF3TR6f4d8TJPGBv57ZNdhY3yzL94c+9dSasS2aEUuR1qcNwMVEgsPDe9L1qQTAmmEEHNOwNiE0nAPNJIoM5pOlHqDEzk9aOT1pBYQ0Ed6aEncDx3oJz3pNdR2Gk80mcikAoFByTSb1E2ITR9aY1qHWkJpLsxsMjNGc0xoTNLu4qXqJCZPrSk0BuFJkjrRoOwZo3UImwbuKSpsx7CEY5pMmq3GBo980mDA9aRue9K+ohGB60hqZMbeg1qjYUIEiF1qtKlZyKRUnXJqnKnBzXLN6G0drFOVAMmq7IT2rllrua62EUYPSnY55rO2pQ5EJNXraDisandES2LSrin+9c6loZD+tGM1lJK12IcFNO2+tRYY6Jcmp1AFVfsQ0SKuaeB2ppPqQyRY+Oad5WDRKHNsQKYqjkXFXSpaiuRE4rK1m9WKI/N2rRvQ6KC5po871a5MszMOp71lu5xnvWFNH1EIpRRC75ph56810IckRk80mOcmrJbDOKcmWOc0/MxnormxpakkGt6BPl5rzcRK8rHzuLtzNjbmUKp56VnI/nTevNOhG2pwx3N2whJAq/5JxnFJvmYmghtmaQcVqR2exckVvCSSMJu7JYTsapy5biolLTc1gralm2jzWpAoVea0pp2sKb11HsOaRchqiUVzXBFuJsgZNWUbjk1tDTYloTYHcVK8YRK1jG5NjKvHYMc5rFvbjBOTXJO6i7mkdyG2/eNVtodq1jTV4nZDTQhdKgkSixuiu6VC61D0d2h3InTNRMmBRFfZQ2QutRMnGTVJO9h3IytRlfWreiJuNwR0pd7dzThK+rG3Ya5PeoHBznNdUUnqZ82twWZwfarkF30yazna+hpCV0XFugBVe5ugc88Vyygtbg9NTNeQsxOaAT61q04rQls0NOuhGwBNdHZXAkxiuHEU1o/6/MrdGrGm9KqXFud1EYuMTNOzJrWAnnFaUMe0VztXfqKROgOc1oafJtcZNbwbWxMrdDoLbBAPUkVYx3rvgtTIlUZXimvHXdCN4kMZinBc1vGNkK7EYY71m6k/yGtorQaZzM7BnJHeoWc54qW+h0x2I2YkVC7e9Q+5USMvUbk9TRF6AyCRqhzmhytuPoMIyetATJoirsEyQKqimO/NbxjYSuIqlj1q5a2rSEf1rqpwe4pNHRaPohldePfNd/4f8ObdrFeOp/T3r18PDqcNWZ2VrbxWMKknoAc/lXGfEHx9Dolqyq6mXHyqp711TmqauzGEbu585eMdfuNZuJbq4kLFui+grh2uP3hz61xwblK50Ql0Zq6fGJEBPNFzC0chPas8QraoUXvcYkwXvzSPMWGM1zNK5rG5WcbjmoGTJ60SVhu9rFiAkEAHirbrhMt3rFpPYV7aFVgCajfGD2rSasrDSuzzNosVETzya7YabjWog4Palzx1zRdWKtpqApwcjiiUuhDaHCTB60pc5zmr82NpWHpMQfWr1vdbSOamSVjRdi8su7096mRgTyK5lBX1Ja6E48sLnNPW5C8dKu1mEIu1iKe53GqpfNQ5e8NRsIZcjmkBGaid2y4qzJgwK54qNualJIUojBH3zijAT0zRzdAcSNyPWoic81rF6GfK+gi9etSpuxyabYwI3HJ60N0605JCSIyMHmnwkhuKa1Kt1NbStbmsZVO/5c9K9E8NeLkuFALg465NbppA431O407VUnjzuH55rVilDAH1rRpNXJv0LCvThzWSdg2HbuKOMZpvQFcaBx1pCMdaT3DyGjg0UMN9RCKB70rjE6nNGT3pi8hKOtS2NX3EIHQ0mCKS2sMMH1o79aYrCUv1qXtoAn40h69aBhmg0MLCE/jQMnrR0GwzR9alCSDd70Z9abGAY0UCEPFAORzSvoMM0E5qk0FgIGKTbxmpuHUPemnJFK9wEzjrRjAoY2tBrHmmHJpWGRuKryDrUtCRVljJPWqkyYNc9SKvY1uU5kzVcoRXJI3i+4m32pCv41g+47lm0t2Zgavom0cisKjSM5u4YzSrXI7pkEg9acM0NK2omPAx3pwBpPYLj1WplGMVnsrEt3JFFSKuTmtt9iGyZVp4HNaRV2Qx4TjOKrT8VvKCWwluUrmXy0LGuM8TaoDlVbqK5qrS0R6eBp3nc42eQuxzzk1A3A5pI+htsQscnrTGB9a1QnuIc9qT61aMZdxOvY0+EYfGM0GM3odDpCZ7VrvJ5aV5Vb4z5zE/EzHvrssdoNTaZAWIJ6mumOiucljqdPt/lFa0NhvFRCL1aIkyxFYBDnFTyKqpRSW7ZLKmMvVmGPNSrOWpoloaNvDgA1cj962U0nYhoViMU1eTVvUIosRnkVaXmqp7aie5LGMc02e6CgjNbpcqJtqZN7OuCS1cxqF2GkIU1zYuS5LmkIl3SELLuNXrhtgrmhG1LmOlblZTu5psiZoSTd0dMSrKpHNU5Zdp5NS+zLVmM80P0pHXNZwXvaA9CGRCKjdcVo/i1EiJkHUmoypqpavQBhXFNKnrVqOlxbhtJ61FItbx2Ja6kLDFICV6USWtyUKszDvTWct1NJu7L5tBB1pw561jLazDcehwetbGk3hV1BNQ9Vyj3VjrLGXfGKfIu40OHu6mTRZtIhjpVgjaeK5KkNibEsa56mpon2SDBq2lCI1ub+nSMVGeeB1q6CTzXVRVkZyVyWN/enls16NJpokZjJ608KM10x20IaGyICDxWHqjbc81T0dionPzDB9BUDnFQ1qb20InaoXOeM1DZaIzkUwmmo6D6ETj86j8s0nG+rHfQTGKMgDitYQs0AxmJHFCxljk11xiiG7F21tWkcDFdLo+itIynHpXdQo6GE5WO98O+HRGVJTOR0+uK7G2ij0+JSQBgdfSvUjCyOJu7OH+IvxEg0K2dRIDM3CqDzXgGv8AiK71e8kuLiQkueOfu1x4yom1TRvTjYw7mL7QmB1rm7yxeGQtg4FY0m0xLRk+mXZVwGrTvvnh3Ac111knEdrO5jncr81KrZHWvOlrobpJajZchMrVYlyfakpJ7hdbk8DlTuqw87P1NLlVrhu7sjzlsUs6AISPSobd0B5uv7xM8VBJBjnFd1uo35EezHWm+ppKDZKTYY70h45NNRbE4jgO9AOWq7Ny1Gtx+3uachbOBSd7XH1uWI7po+9TrfbR97mlylruSx6gepbNTpcB+c5qZrW4lvoKWJ6UnIPTrXM9GVbTUCOacqcZ4qW3YpaK4u45waXoaajYJPqGO5ph5o5OpDn0I3FRsvviritBO4AHrSk7fWm9ybMGJPNNL+tC1Viug0OCeamiPFOKs9RMftzzVix1KWykBVuPSrT965Mb7M73wz4vDbVZxnpgmvQNJ1tblRyK2imOUbaM27e5385qyrk96cooEOU04HnnmpHcXikP1oTFqNxSEE9KTBjaTnNIoPc0hz1NArgB3yaQkih6vUYnvR1p2BIMkc0nNJjYmeaU9aWxNg96OppXKYnegmjqAdqaam/YaEBoJ9aqwB70Z96h3ACeaN9CCwhbNKDRYALE0dTRsMCeaM0JCYmaaSaEgSDHekJNDD1EzimtzStrcGRt+tROo5qVfUpFeRaqSoSDWE0WVJY/Wq7Jk9DXNOOptFiMntSxW+89M1zNaA5WL8cQiFOzXJMzExmngVzNN6CHDmpEFJq/ujY8rT1WpUWtybjwPWpVWklbQjqSqD3qQCto6bCZIOBSg81tTs5EMezgLVGeQsetbVHaNxwMrVpNkJOTXnurs01yx9DXDUb0bPby3RNmXIpzyKhZDnPrTiz1WxvknJwOaa1ux7GtFIxlNCPbMOmajMBFaJu1zndQPKwKdCP3gFJ7GUpXTOm0mLbGDil1S6ESHmvM0c7M8CvK82Y0J8+XJ9a6LTYQoHFdU1rZHM5W0Om0xeQa6C1hBqoJ8tibaXJLhQgrLlmJas5OyshRHwpu5NXYI+RWcVqmaXL8YqRfrW0fQzYyYleaSGXc2DVRbvZjizQto/MNWJEKc11+ztG6Jk9bDDeKicnmsu7vxuPNROWlxxiYmpXpCnDYrCR2mm6968qvKXM+xtFdTqNIjxEMDnFWrm2LrXbGD9kkhSlYh+ymMUx48Ckqdlsb053RUuIyQayr2B+1Zzp2V/6/I6IOxWgVw3NXQvFKMbmk+6GNHmoniNKUdbMhMYYqY0VJ35tENkLx7absJGa0u7ksTZxUciVtZMlsruMHpTT05qrOwMZjnNIRmsr8tmwt1FApfxrGpFyBD1HIOatWLkTL9aUYtvUfc7LSOYhk1omPNbTS0TMvQnhG0VITk5zXLVjdqw7D1bC1Lb/NIDjvVuPNFREnodPpUIIXPTp/Kr8sO2Mkdq740+hhcpiQhqlV8jrRQk7tFtCFwDnNOE3Q13Qk9CWhzOWUmsTVxhSR1Pv9K3UbsUXqYEpzzVaTvWclqboibJqKsdjQRgQM1E2c0172oXEIyPao2xWqWor2InYdqiwW71tFO2o+pJFFuIrRs7AyMOtdVON2TNnUaLoJlYYXoc/55rvNA8Or5a5HUjn16e9erRp6HFVnd2OoRItNi3lgMD6enfNcF8RPiXDo9s8UTgykYVVPT9a2qTVOPNIyj5ngOt+IrrV71p7iXezds9Ky55CwzXjyfM+ZnQPtmHcU3V7VDaMwHNdEEnYmaaRx53RTEjsa1rW7LxDca6o7NMN4lW4OZDSIx9a45RRSuPDDGDzTGQE1lOGjsACNqcM9O9TF6WZaatcdsPXJo8wjg0Rjdh0PNLeYKcmpmkV265zXWry1ZTT6A1tuUkVWlhKjnrTjKzsQm76iJH781KbcYzS1u2h36DGj46YpqLluRVxv1BFmG1aVu9WV00qpOCKTbSJctCpcWzRk1EcgUc2hV9BFc+vFWrecggc81TV9C4PU0Y1LJmnhcda4ZvdFSegu7Jo3Ac1HK0KGu4jcnNAbtzTTexUh+cjFMJx2obYrDSu7mmMvbHNOLeyFdoVYwB6U1lAzzVttsV3uMIwKibPeriFxAp9KnQlRSdyW7oeXwOtNK/xUXEyW2u2tnDISDXYeG/GXluqNIcj1Na05vlu9y73Wp6PoniJLhFw2cj+9XR2t4JV61snfUnlLqMCOtOJxU9dB2HBsj1pR71L0GBBxTCppaAN6Gk75oYmNJo5NC21KSCk7UgDg0negVwPrSUDEJpR60ncExCTQDRy6AgJGaaaXmDDOKM+9CHYTOe9IQKSC4dO9HSmNCDpQCCaS6gwNH40rhcA2KA1PcQZ55oBFSNoTOTS5oQkNamnincYme5oPTrSluHQacVFIPSpje47kLpUEiHNYy31KvcrPFzULQg1hUXkWmQvHzip7dQo9a5aitErcmY5puK4Gm0IXHNPXNZW01FfQXaakSs9UxbkqjJ5qVVos2SyRVqRUq1HQkkVaORVwTS1E+48ZPenZxW0Y63JeoySTjmqUrDd1qautrDj5Gbq5DwkVxl7ZZkY46mpqU+bT+vyPWwc+ValNtOPUiom0/jOKwlFx/r/gHX7cja1CnFPS1GKcVdXZlOo2iKeACqzxY7V0xh7tjBz6ELRc1JaWhlnHFZTdkTKpozpoYfs8GfasDVpyZSM1wUE5TuzxqjuyPTELyiuv063+QcV2zb6HPJa6nQadDjBNdDZrlRShF31FfSwzUlIWsgREtzWOIupWsOBbijwuMVbgO0jNENGXLUvKoYU4DZzXXGN2R5Fe6mA4qK1JZ6JW5tBxRrW8nljk8Ul/f4iIBrpU7QaYktTEn1A4zmsu41A7iSa8/wBo5LVmtkZ97d+YOtR6bHvlBI6muKV5Tux7HaaTBtQCr80QAzXs04pxMZ7laRQRVOQAGq5E2b0H0IJEyPeqk1vu5NZVKR0plY2oHammPFYRi07mnMMYHHSo2Q96ycW3qNMZj2pCgIoGRtGD2qMxetJXtcTXUY0eaikjI57V0LaxlK5XkHNQsPzqo31uNdhrLz1NJjtSaTYPQXbRjFZOL6DHD2qxbDDgn1qZe6gOw0N/k45rYUHNaS96N2ZtJMlUZNOIzWMld2DmFXnirFv8rg0oxd9BHSaVL8q89Pf6VozSZjwD1ruhzWMWZ7jDcmpE6Uqad2XfQCM96ADnrXdAhslDYU5rC1iXcxBP4iuiOooPUxZTzzVdzk1EoaXN0RN0PNRNWbXQtMYScUxqdlbQpaEbNgdageStIe8D3GjJqWGBmYZGa2imwlJWNSx00yuABXWaFoBdlyuen9K9ChSOWpM7vQvD3lgfKRggNz9K6Fnh02I7jt4zx6/nXqJKySORu55Z8S/izDpEb29q4eY8AbuleFajrtxqk7T3EjO7HqTXn4qpzvlT0R0QjZGWZiZT6mpWJ289a43e1mU0rCxbsjmrpUzRbTzmt6PZkSd1c5/VtL8liwwBVK3IDYrqg+hKd1YluIWYZUcmo44WB+asK/uuxfNZaDpCq0iMHNYu6sU1oXIYlcdcUTQpEM5yTUWb2ROt7FdpBUcrErnFHkzSSSVjyrfg9fwp6TYNdjbRVrl63uwRyeKlk2yDPaplHqDj1IFjw2akVlJxnNX0Gl2EeMke1NVAD1qU76ILXRoWG1W5/OtBnXb61V9CXDqU7mEPzis64t8GptZXYKJV2bTnNPjfD5q+ZXDlL0V8IlxmrMVyk3fBqJU0/eNErq6Jwi+tG3PSuWad7Diu4MmB1pFGD3zUXexXL0FZD2pm0jrTvpyg1YXaTRs/WqW9iGhzRgd6hkXFJN3syXoyM9aj281omPQcsfcmldgq9ead+xLauRByW5NK8xAx2qim7jRzyalhkaM5BxTvZE31Ok0HxRLaOFdzj1NekeH/ABWlwgBfitVK/Uc97nW2WpJKv3s5rQjl3itRLYmU4FO3e1ZTuFxdxNBpNJD8hpGKaw5oExpB60ZwKTKWomcUvvSs9wGdeaSn0CwvakJqUJgetIfXNF9RhmkP1pAJjPNB+tNWDzE5pOpofcpbCkUhqVqKwfjScYo16juHGOaT3FIH3AijPFFw8wJpATSC4UFuKAeom7IpQQRTegNW2ELY5oyDmp8wY08ikPSmNCZppA7Gp2EQuPxqJxgUpRuWiB0phiGOa5pu1ytivLHls9KRDt4rlrLSyKUiRTmn4rhdw8hOlPQ1m1zbiaHd6kRM1g3yuwrlhImPOKkRT3qop2uS32JVX1qVUp6kXHbKTZzWlm0JjglDDFbQutCSCc7RVGdsVjNamkDMvW3g81kzRDdzitErq52U3oVpIwaryQ1zyVmbOWhTlgy2cGlCbRz2q3G0Rc1ytcY696qsM9a0gnbUia1IzHuNbmg6WZSDjNc9aL2RhWdomzqWn+TbEkYwOlcPqilZie1Z06UoS1/r8Dzn3JNIYeaAa7nSkDRgitX8VjFo3rWMcYrZsUwQM1tBO90JLQkv4gyZPOKyxD8xrHFXcioaFhIuM00na9ZWGmrl+2lyBk064mVVPNdUGlC4NamPcXO5+DVvTmyQe1c1Obc9S2tDRmfC5z2rG1C8AOCxz9a1qSaTRnHVmc829TzWJf3LI5Ga4Kl0ro1VtirHMZW2muh0e3Iw1Zpucrl6HUWJ2KMnmp7q6VR1Fe1TnaFzCa1M2W/UZyaqNeBmzmqjVXMv6/U0pKw4TbhmmM2/rXW4xZ1JakLDmmMua4qkdSkRulROlc9SPYvoRlOaUx+tYcsuoLYTyyBmmNF3NPVaFXQ0xjFRvHkH3rWnFtsixUngxVZ4ttaRuQnrYiKHOaQrzzWc073KsBWgDJ5o5mJMeoFSRkhqznfoCep1Gh3GAFz1FdHD84znNaX0uiJWJ1A70bfWlOP2jMMYNTRHmsYt9Sro2tMkIOOtahlyvTNejTu4mehXcFjUkanvV0oNibHiP6mneXtGa6oJ2Jb0K93NsQ81z13Kzscnqa2a0KgluUZDzVaQ4qNWrM2iRE5qM81D3KGsQO9V5ZOetJRe5S7kR5pu0tW0EN2tcnhti7CtjT9JaQg471204djGpJHY6J4dxtYj05xXdaNoSRIpIx3z+VepSps4akkaV5fw6XBvZkAUdd3WvGfib8YBCHtbCbdJghnU5C/rRXqKnHQqkurPEdS1Sa/mMkjs7Mckmq244rzFdKzN3K+w6Ic7qm37uKh72M2rPUb5oU4FX7F95HOa2pu0ipRViHXoswnA7da5i2GJ856V0Lm5r9CUvdZsqFkTKjP4VRu8xGliKetxQV9LlGWQuetLA3PWuR3aNorWyLiXHljvSyS+ZziiLcYjUdSq5OTTWPyHJ7VNm2TLXQ8sKDdk5FOXiuyTbQXehIjAGrUVxikr9yrX0HNMCtQmUg5/lTTS1H8LJlugRg00T7j0ppRTvcbfYtQSYGcmrUdwQeRU3DdWLSkSCoLi23AkDNEnoTLmRlTxsjHioB1pqOxL02ELHdzUscxTnGK036lKWmhYW/fHU1ctL0MQGqHBM0vc0Bhx2p8cSnqOa5/Zq9yJTsNkAXtioyQR6UpRSV2KMr6iqRimOCelTJfaL5hMnFMf5hms731H5kJXmljjJcCqvdWM7FtrMJHuqhMnOM1VNtslXuRBQDRjPWtWtNB3FC5OKkVdgzik9NBtMVSSc1qaXrE9jLu3HFEGouzKg+56B4b8Xh0Xcev+1Xc6ZrCTqCW5rqjUTRLVjaiuA65BBqZW9aqyF0H5A5NIOuazt1KQ7GaRgO/NRrewWGkU00xoac0h4oT0C4hyOtHFSGoGk5oGJ15oxTAMU3rS8gE5peozmk9EAhzQOKTYxDnOaGGR1ofkAh45oNKV2O+ogyKDnNSu4w6Ue5oJAmm5oaH1CkPFK4dRM0tPXqAmc9aAc0NBe4hpCaSuwEY03Hemkht6DStROlRKWoJkbLjrTWTisJopPUheH2qJosE4rKcbhcFXAp3bmvPqLllqO9xQuaeq1i9Rsfsqe2jy4rBK7sDehvWunB4t2Biobix8rnFejUw6hTulqYqfQrhOalVa4oyTYMk2cU0qKtxdiXoGKRhtGa0g9NRlKds8VSuORXPL4rlxMq5zuqlIMNW8dkjrjsVpByaryCsWrlpkEg5yahlPHFD1VxrYpzVWK88mrjK1mS2SW8W+UA12/hazzhvSs781RJHFiZ2Rs61aILUnHJFeWa/GUmbPrXTiIaI409EZtjceVMCa7nQr8PGoz+tY2d9/6+8Jq51+msDgkVqpOsfORW11FJXM4xuR3F2GU4PWqySDua560ry0NVEf9qC9DUEs+WzmkrOI+VIkhuiDTrqcsnWpb921yHEy2cmT8a1dOYADJrGnqxy2J726VIy2a5q/u9z9aK9ToKJALnKdazLwGVzXJUd1Yq2pNplgzuCQetdVZw+UvTFb0Kelxp3Li3JTpVHUL9ghIPNbyqJKw2upkSX5Y9aal0xPWso1FdSv/X3jirM0LabcKnLZr2qc1KPMapsjJweacDmpnFS1NEhrAE0wxgnmsJU7FXdhBEM5pWj9qxcHbUVyMrimsuahw6lKw0pTGQVUItK4r9yrcICTVWSKhNgu5AyEUwoO5omrAxpSlC5rKWwmP24pygCoSdtRXNjRZGVwM9TXYWXMYJ61vB+4TO3QsA4pw96XNq0yGBAzUsS5PSso6ysga0NfT4yMMfrWicha9NRaiZLUVFPepVXFa01oJk0Y5Ge1JcFUUnPFdEIkMxNQn+XrgmsaZskk05bmsUyrKcdarynmspNo2TIiaidsVK3HYgdzng1GctVvXUpaAkZJxirdtZl2Hy8V006dyJNm5pmjeaQdhz145rttD8NfdyvUjP6V6lKlZHJVnfQ7XTdIS1UOQBjH8hTNY1220m1Z2dUVRnGeT0967PhRgldnh3xE+J8uoPJb2UpWPOC4PNeU3cklw5YksT1J715dWpzyb6I36CRWO+Pc3FRyRxx/Ke1cbk29Cop7oQDIzUb57Vol71xJa3ZF5bk5FammIVIJrSMrPcrmWxb1G3E0XPauKvQYrkhVxzXXC+jRnDubukMrQjdycc5puq2oYZArSqk1cS0kYE6bDiiFG69q4XKzubJk27ipIyc1m029S5PQc0eeQM1GYsDmqmrLQlPueT55zS7ua3d7pmm4/IHXrQDj2oV9SWrIduLdzxTS3HXkUO8QV7WDeSOaFOKdnsCT6ksU7D1qcXjHvVtW0Y2y/aXA7tmrTXq7TSs2VytmbdvuJI71TwetO2hFn1GbSDzTg3Sj1Id90SDnpTkcoSeaTvexcFqXba+K43McVcj1AZ4zV2vuU46sk+1BzyaNymsKq00CK7EkbL+NEgB71haQ3GzGBC3elSLJANRytg2ycWIIyTVadPKPXmtIw6kXsRvduwxzURwetPW+hTRGV5zSqozVu7WgltcUJSkepqEmw5rFm0gWQ5PSnXcOwdgKHHXUht3RHaX0lo4Ksciux8PeM2jZUdiMe+a6KUro1aclc9E0XxLFcIuXz9DXR2t+sqghq2V3qZ2tuXFcMMk0/NS7juGTS5z3qWVcOKaRml6g9hp5pppMEugjUm40krgJmjgUeQwpCOaExIKax5o6hcQ9aKXkNATRnIpAhCaM0IeohNJ9aADNGOakNtQJxyaFJxQ9gEPPekzzS8iwPNJip20EtRDkdBSD1PWr6CkITzQDzSsyugbjmk59etLyARqQZJ603tcfQQ4Jpp61nJCGMuaZt5rOXYdxNu7tio3jrF7kkJTnpS7RiuerG6uOIAYNPHWuKUXsW2TItTQ/IwNZJWkJs3LLUFjTDHnH0pt1dLJ9a7qmI5qdmzLlaKYOWqZFHeuCnrqNkg5oZeM1uk2rMhkRODzUM8mBQ/djYaRQkfLVHIu4Vyt3uaLQz7mLAJ4rOdOua6KN2jaMitKtV5FA6ilK/Q1RXkGTVeZgMis7O9ik+hRlYVEPmrZRdibl3T4dzg12eiXQtEA49a41U5ayR5+IV2P1nWkMJUEEng8157rrCRy3XNdeIqc89DG1loc+Hw/411Hhu6JkVc96TbjqSleJ6Fp037oHJNTSXuCfmp19LWf9feOEepWbUAW+9ThfAj71cbfvXv8A195s46ET3pU9ab9tHXNbQldf1/mJoamofvPvfrVxbnzY+TWd1Zr+vzIcSvI2DnNTQ6js6tWMJcj1DlIb7UvMTAORWRLOSetRWd3zAlYWEs/FWIrHe2SKyfvtA2a1jbrGo9avlxtwK9KnGMIWI8yvJNt71n30xZSea46ydrGm5jvNhjk04XOOc1EI2tYGmWLfUVQ4LVpQXquvBr2sNNNct/6+82inYl37qTcfWtmWthDIetJ5lJxV2UO80UGTPOaiURWGkgmmluaz5NShpbvmoJZcHGalRvohpFd+TUbDis+VjsyNo89eahaPFKXmKQwgdKAtRKLSsZt6WHbeacBk4qLMaWpo6WCJR9a7OwJaIH1Fb043WpnLct44pRWNRWbRI9QDVm3Q7h3oo0nzXYXe7NuxTgE1cZd54FerZuKRiKsW35iaXcDTitATJFOKqX02Bwa64bEpXZhXchd9xaqEr859Klq5tYqzNmqr8nk1m49y0MZqglc+lRya2uaIiCkmpEhL10Rp2syWX7PTWkP3ck+9dLo/h9pGBAPP616FGldXZhVnY7XRPDQTaxXPb/PNdVa2cVjESxGc8fp716NONlqckrt2MLxT41tdFtmd5VG0fdJ/+vXgnjb4lXOvTNFFIyQE5GD1rmxNa3urc1hA4q5uC5LnjPvVeOYFvWvObNLWHT3hVMLxVPeXOSTk0oR1Fdkqu2dvWlnPlxg+taRjoDdhLeUYGau20pDDbSilzDcb6s0WPmRkZ7da5jW7A+ZvHP0rqbvaxD0K2n3JhbafyrY4mi3Ma1lJyjYl+Rz+oR7JaYjYXg1yWs9TWOxIAepoJOaHbdFPRkqk7OeaR22rzSavuTZtanlIQdRTGArdx6s1TYigKc55p5Pc80uVvUUn1HJ0JPemkAZ4oab0Qm9NAX3pQy9qdr7C1auFG4g1Si7XJvqT29xsPpVr7QrDGaqSdrmjuKQrY70PCAuazV2wd+pWfg9Ki6npVuFyUieMDPWrC2+4UrO12CfUZJCVHApoaRT3qtbB5sek7DntViO5PbpU27miki1FNkjkGrQfjpWLgugNj4mXPPWh7hA3HWoW42mH2wjgVUmmMrZJ6U3LsJxIypPIpuMDk0oaEvsN3e9KCM5p8rvYpLQcDx0pNwzVRXQnlJFuvLOBSSzNL3onsD2IsdzTlkZDlTg0ruLTQ4vqbmieJ57JwryMcepr0Xw/4vSYKN3J5610U5a2Y5K+qOzsNWjnUENWnFcK/etbXM7WJlbdS9Kyki0ITil9O1QwG8dTSEYOaVh9BjD1ppFMYhpCMVL0HfQU+tNOSc0JkrzDtRgZ60wEyPrSYNLqMQmkz2p8oIXHFJjNS9EID9aQ8rUt9x7MTAx70oJFA73A9M9aaOn1oGgwaM0lqwbEOc0mSfpTZQjGk5PNLzExSKYfahMaF+tIT6Um2kAfWgVN+4g4zimsopNhsxtNI+lS1rcaGd6RgT1rOUQ6jTFkc03y8CsOW+jENKGhAQa5p09R31LKj2pyjmuWUWnYCWNmqbLN3rOTdrDuh8Y5qdeahabEMkXk1Mse4V6NGOlmQyvcKEyTWbct61hVjYcWVHPOaaSTWDSWhd9SC4XcOlZs6YJNVSk9jSL6FSRc9aqyA5pyXvWNrkEmADVGfjnHWsru9ytilKwLU1clsV0paGctNzY02PIBNaEtwYE4Jrx6zcJ3RwVJamLeag7scms+6YSrnqa2g5SVyXrsYk8JSStDSLkwSK3vXddct2OB2Fj4gCxbd1LNrm7o1KpO6Wv9febRp6f1/kVH1k7vvVYt9WLdWriu09X/AF95rypqzLP2veODUMt02etOFRx0M+UZHcPv5JNa9rdEIM1Dqau7/r7xuDaElnYmkj3MawlNS2IdNodJESKqND81JRv1MXe5YtY1Bya0Y3UCt6cUkkwcdB32xVOM0x9UCd66oy7/ANfiRYqTaqC3Wq8t8rqRkVNSKbf9fqaxj1ZlXNwA55qu123rXMrwehSGC5bOc1paffMCMk1pCs4vmT/r7x8zNhL07c5oS8DN96u2OK5mlf8Ar7yossLLuFIxIrrTNIoaG9advJpy7j8xC560xpetR0KGGQjvTG5Oe9JPXQL2G7T1ppXrmueaY7uwjLSMnpUyV9BW7kLxbecVERk0WbJsKQelOVSDSa0JtqaGnPiRfY5rsNNfMI9hW1L4bGc9Gi7nIoXk1w1E3JjfYmjBNX7SPJzXVQTb1M57Gxax8AdKuKAqc/rXppaamUtSOWXsDUYJJqXurDSJGkCr+FZWoXGQQD1rqWwkZUzVTmald7GqK8vTNV3NZmiIm9absLmmoXdyrj47cs3Staw0pnYce+a7qdO+5jKVjq9I8Os5UsvPqK7bR/D6xKpI6dT/AJNenShZaHHObe5rmSHT4SNyjaM9cGuB8dfE610WB1WXdLjCqrd6uc1CLkwitTwTxZ4x1DxBdNJcTMY88JngVzLzOz5JJryqjTdzfZCNKzrtBNEUb9j9c1Fu4+bTUbcv5fuaLMNO3yirjHmZL21L5sjEN7Gqdwxl47CkvhYk9bjYjsPIrQt2wM1lZp3NHoXIps0zUo1a3OMEmu2nC6tIzl3ORuV8mUkE1dsrzeuCxrog3ZjtdNorXqgsTVJpfLOOtclaLvqEW7WJ4ZCwzTwctzWbeiSLZOGG3iq8x4zmnMmzueVg84oA55PFauTsavshcDHajIAznNNykJq45CAM5odgRT5r6hZ3IyfejcM9aXMwd9hc55zS7gepou7NCUXeyAGnKxHerUr6Md2SpMw/iq7HchlwTzRa49yN038immAk5pc1kKSARkHOauQONo5pqV1YqOqJ9qkZqJ4lJzmodybMRLTeSc8CoJ1MVCfQfqMjuTG3WtG2vlkUZq2rltc2pbQhx1qOZMDrXNUi4vQavsyHJXkmnRKGbtzUXa1YNu5eFkTGDxiqU8exjVx22Mle9iDaSc05RzyaOcvYV8jvUeM96SYRBUO7k1NsAobYmI44qMqeuapE9RyAgcmrdlqUtnIGSQgj3ojVd3YaZ2/h7xnt2RyMAfXNd9o/iKK4QEuOe9dMJN63E072Z0VreJLjaw5561aEm7pVPXcEKBkUlZlLUXBxSGjcENxSEY5qdR3sMI7001Kv1ASg000FgNIPrSbtsAvApMVKuw1sIQByaSrFdh2pKl6lCUc+tJhcCKMetCYBgUjc1Ld2O4h470h5oTGgPApNwPFCd9wQHHrSZGcUmmOzEIOeTSGkMaeaUUN6DDNBz3NS/MkTNJmiWgxDzzmkPFJvSyJQ2krN+Zdh3WmlayasyUhpTJpAuKTimQ7okQe9SAVx1qbvoUmTRJUwSuaUQHAY5p6NU8qTJuSpU4faOtdlF6iZTu5c55rMmOSc1jiFroNaFWVigzVcXBLcmuWN73ZoldkrDK59ao3MeM+tVGVpaDW5nyriqsgxWst7s3uV5OmTWdc9TyaXoJSKbZzRENzD3o5rK6FLa5vWCbEBpNQkO2vGqtyqNnnS1ZgTsdxNR7wRkGuuKaSuUkVJ4d5zTYo9tbqXujs0Sid04DVKLtiOGNS22jamxUd5DirtukuRk1jN21OqMbmrbAgDJq0IQ1cdSt7vmbww19SSO3GcgVYQcda4pVpanVHDD1XNTQsFPNRGo1K7IqYZcrsPnlVU61mvcjeeRXpxkrps8edOzsC3gXqak/tIY+9WsZrX+v1IexVm1Ign5qpTaixOSxrXn0s/6/EzK0uoMe/606G+J4pSm7aP+vvKv0Q6Rd5zUbRsO1c8ZN7i5riBSDz1q7aNtGTUu7egXLhuCVxmn225nzmtKa5ZJtlLQ1YcqozTjz1r247XRvF6CGl/GndtFdBMjHJzTCamzbuxjCuTQq4702rCFK9800qO9Q4toa1G7c80EE1moDa1Gsuc1CyDHvSlvZEtEe05p4GKJO7SRKLFsxVxXVaTPuUAnqM1cbt8qM5s2IzkYp4X0Fc1SDTsJ7k8KDIrStl24/zmuqkmZyd9zTgm4AqV5gVrrbdrIl6EBYlqcrbee9TF9BNkNzcbe/1rHuJiScmutbDjsU5H5qvI2fam4mqRXkyahKE5qF5lrYFhLGrVvZNIeldNOF3qS5M3NN0NnYfL1x3/APr11+ieGdrqSM5Pfj+tejQp9jlqz6I6/TdKS1TkBRnnn/69N1TXINNjbdIowOuen612JJIxV27Hkfj34tBXe1sW3P8A3gen45rx7VL+5vrh555S7sck5rz69T2krdDdrlSRkyl5Gxyc1DNbPGmTXKxSlbQWyQMcNWpDaqRkVMpFWbRR1G0OPUmiyZbNdzYyKcJcuiGlfQfcaiJjtB4qvnkmrk+wOyFiG98k8CrqDAGTUPRk9SVJRTpG3xn1rphJoV0c3q0RDkd6o28zRNg1vTbKvpZF51Dxbj1rKnQo+SetTVi7Cg9LE0EvFSFvU1xvR6Gtr6DlfNOMbOuWpT5r3ZK3PKFBU5zmncnqK6r3K31HAAimnjnrVO9gW9hPMJNIWJPNZvZMJbgRngDpTQrHrTTBuwrAoKByM9KItrUm7HbyfwoBJ6mn8IbMduPWlEjA9atT1uVdJFqCYj7xFWCcjPHNNyUtUN2sRu4HWnI5xxSWgRtuO+1OOvNSwSlmyaUl0G7dC5G+3k4qtc4kJJFKSVvMhxvsym8OOaYm9DmrjLS7HGdmWINQkj4arceoLJxmiceZP+v1NL9USZV6s21uW+YDFcUnbRicnsXhdeVHtas66k3tkAVo7ctx+ZCvPbml8sjk1lHYljW5BOabnIrW2oPYkTaB70biaUhLVCHrTelPUS3F60m0de9Te2w2PSR4m3A81vaP4qmsCqsxIz3Ga2jU0QJc2h6D4f8AGSTBQXFdlp+sJOoZX3V0r3tyWrGpFOHGQeKlAB5zUzQ722A5xyaTPPFZWsUnoIeKDzRYY1l/OmEYpboYjDHNJg4qHsJCHimnrT31GByDzQT71PoDEJJ5oxTvYQmMUZ5p76jEINLmpbExD60ZJ60kPoJ1pAPWn1uMUik+tRdt2DoJ2o6URXcEMIOc0Hgc1T7FXEznrSHJ71HXUGHXFISaVwDNISTQwQ08d6DS6DuJkGjjGc0tbXQajQQTyaUD86iV7DuKOppQPWsmmJsdsyM00pijbUh6oQKc5qVRUyXMTsWIhmpiMVw1ElcoazcUL65rnXvSugtoSq+KQ3GOldFNcpJXmbeaqSJSlFsaZXmh3LVVbVg+SKwlEtMlcECq02MGpvdlruZs4+Y1SlXBNbadTRbleRcis+6BBPpWduiG2UXOCals0Mkg9jTmlGNzOb0N+IbIfSsu+nO8jPFeM3epdHFbUypm5zmq5bBxmu6KbRUVYkC7hTTG3YUuYvlbFitHlPTirsOmZHIqalSy0OmjQbLltpoXt1q8lqiDkAVwVKzeh69OgSDbG2MU8SnPXNc7V9ztjTUSVXPrinKxxyayaKJA59aUuQck4qLEuNyC7mbZw2aymkYNya7aUm0rs8DGRtIbLOwHBqq162etdNO55zFSdnPNOZGYVpJ6XZD3IGt2zVi3tm9Klv3bhcvRRdjVlLUOORS5b6lNdiKWwO7gVLDp0mBxUxi7/wBf5E3V7kwsXDc1dtrbYOnNXD352ZpGV2WyMAc01iO1e7GGh0REA5znmlJxTavsUmNPFIenrSs2NigUEDFRFO9mMCOORTQCepov0GhpHrSYyfSpULE+Y0jHJ7UwkMelHJfUVr6jGTJzQAQealrlRLQ+M/MCa3NHuSHHpTptp3QpRVrnSW0gYLz1q6q5GamcdTGRNCORWhD09auluySwGx+NPVie9W5O4mKG9abJMFXriuiCvqiehm3NwW4qjM5HU8V1KNkVEqPJ3zUTnNOVzVEe3PWnx25aqjG71E2X7PSmmYce2BXSaT4dYsp2+1d9GlfUxnPszstI8OpEq7l4JAP0roI44LGEsxAVevP/ANeu9RsjlbuzmfFPjyz0S3d5J1UKOBnk/rXhPjn4pXuuSvFbSNHATgnPJ/WuXFVWlyo1pxtqziH1IuSWbryab9rDKSK5VLlibPXUS3dN244zUWp3KlMKafItyZK5Wssk5xXQWajyxwMmueS1aBMgvrcglmNYl3IBlRin7PS6FCTepBbL82TVg5YmnK97laPcfExjznrU8Um/qetUmpOxKsSZCmpogzjGOtXrcS0WpT1Gw3AtjmsCaMwyEkZreHmQpalm3k8xecVDe2rY38VpUs1cakUo1KdalVh1rklHU0UtSe1Cu+WPAqe5lGMDFD1KtY8mBANLtzzWijZXKtoJ/Km4461Tk7WC1hByelKMZzjmhR921xyjcXpzSc560KNtwtfcDnPNJ7kUW7C5OoYHNBB+goa11E4thijJJpxtuTyvW49G+arMcvrRa2xaWl2I75PXmlSXb1qvISQ7zAxFTQyDgilLYvk0uTmXCdah84k89Ke61ISJV2kUyWMHFIq3UrzLioVJU8ZzVp2Wgl7qLUF6YjgnitWz1RAuCeaynSU9TSCvqPmuFc/Kc/SmKhY96wlBrQltomjtWHJBp0kW0cjFJJXJ5b63K74PSoduw0/IaVhA59Kk3EgDNDBRtqxrZzikB5xSTb0K0ew4cdaOgpW1IavqGN1I3HI/Cr1FqWrLUp7RwyORjtmuz8PeNnUhJGIPc+tawbSuaWUkd/o/ihJ1Ub1OfcV0VpqKyjhhXSmpGXLZl5XDjrTlGazlFDTYMuaTGPes2iug3FJikx3Gn5j0pNpFRbSw7DGFNOaaXQEJ1ox60PQAIpDxU6jYnWgg1ei3BBjNIOKnTYA70DrSSGGDnrQSAetJrogG596McUAIc0pBobCw1gRz2puMjNK91cYmCaQjBqWFwpDzmlbUoQ8UEk9qT2uCGnimk1CbYa3E6UhPFUh3AUoJpMaFXnrTxWMiWPGPWnbc+lJ66ENAU70YOelNRIZYgGTVl0+UGuWvELldutC8c1zUoJOxV9BzE4qIj1rpdOwrgVJFROhojC61C5GVx1qFx61zVYPcqOpFJ04qlc9yayjTtqy1sZ1wetVJRnmtZRVjRbFWUVn3Kk5JNS4pbDuU3qWywJM1nOLauTN6M1nmCw81iX024mvNjT1be5xpNme8uTyaaqtI2BXUnpqaxi2zStbA7ctVsWQBriqVNT0qOGvqSxWwU+mKsAqg47Vz1J82h6dOikhPMIOelK0rNyTWdjsSQIxPfNSKMkccmkxtFyOP5eetDcfWsHe5n1BTxk0uc0mOxHOu8cVQmt+ehralKx5GPpq1yCW3OOlUntyDk12U5nitJj4l28mp0O6qk9CGiaOAvzir0Ftgc0RiTYmFsCelW7eAAU1eOgyytorEcVo2unK4HFbRp63/AK/IU9i0+kADJHSs+4thE3HFbQw8VPnS/r7hwk9iuzc033r1U3E7Y7DgO5pCc1N7ANPrijGRmm+4x3bNNzkVnZ3uCQjHFJkZzmm9tCraCZBz60m7aelSk3uSk9hPc00jn2qbNMaWguO9IV3USTI6hs5z2q1ZSlJM5NNLW4PU6bTLjco5rYhfI5pVF2OdlhDVy3fA5rOlzX1JepNvyaliye9dCWorhJKB34qrcXHHXNdVJO4knsUJpOOtVJHzxXVa5SViA880CMsKrl1NE9CzbWDSNytbWnaCzsDjH1//AF1006euhlJ9zqtK8O4KZHPHv/WupsNHigiBkGMf59a9KnBJWOWT10JLzV7fT4x84GBznGP515j45+K1vpaNBbuZZSMYX/8AXTqS5Y+Y4Qd7s8W1zxJf+ILppbyVmGSVXPArE1BzGme9eRPm59epu7GYZsnOcmmm5I7mqaZVtNQS6cE4Jp5DSnJrSU9LEN2J4CUYDFdLpCgrk9TxUU1zN3E1poRa1GUQkfzrkLyby5Oap3Wg4u+gkF0GGM81o24LDrUPUJLSxWvHZG2gnJqbT5D0NVTWgk7I0I1LsM1oRIF+ao6jtZDLtgymuY1RSGPTFdKsveZlYp2swifrmtAyidM4ra6ktSnBlK4iA7YqlJ8nWuaSs9C4j4JCvepZZcr1yaxk2tEW3rc8w2+hzSgleprS7vZhzCEk9KbyDzTvqDYD1zSbiG604tsGSjJFIAQaAuDHHJNN3Ed6YIAevT60pbI/nQxc3Ubk96ADnNCVtAU7qwucHFODtjrTlK1gbFVietKxOM5olJ7opPW4Kx9aXzWHQ9KE31HfWw5blgeWqZbkE5NW9RkiTbjkGnvKSKF2ZLehC+5hnFROCOTSe9ibob3zSiRlPWi7TshptE8F7IjDnIrUstUQt+8FOUeZlJ3Nm2u4ZAMGmXjIRweawqU+V3QnFp6meYyelNaEjqKwcmF7bjGjxwBigKymtFtqJsUIT1bFMKZb1pc2ugm+wpB9eacqkDk046jTFxirFvaGb5scU0wm2kJNbhWpi7ozlDzTjJpkxlbc09N8RXFiQu4ke5Nd14e8bo6qHfB9zW0G+jNW01odvpmvxzgHeK24L1JMAGttzJMshw3ehhnpWcolJjStJ3rN3uUhOnWkxnrUtaljSmTzTGWlfUVxu3AprUMbEJOetByOaHqAnI60e9LcPQDyKTNAB2zRTH5ATnvTTz1qbvqGoY/OgnFF7gJz60ozSYmNJOetIBk0m7LQfQQ9aQikAh4NHOaPNlCE803k1NrrUBuKb3JpWKDJpGFGwuoDilHqahXYdRc85zSg0mtNRoetSqOajXoTIkC5oKd6UbtmTHwna1WmkBXrSqxTVh3IGyTQAT3rnpUrO4dBDSda6pIQhPvUbtz1qGugWuROQO9QOazlC40n1Ks0hGRVKeTtnmsnBWsjZIozEk89KrSnOc9qzcL7GnS5WkB61Tu0JXrWc1qG7M6RCD1zSQuUfJNN7XIdrE815mPGayprgsTXnOPvs5upHHGZHwK2LDTRwWHWs68+WJ6OEpOTuaaRJCuTTHfOTXnXbdz3aUEtxm5v71ISxHBqjWNg5PGcmgbs80DdkSKCDkVagQjk1nN6FaNFgSFPc0gzJyRWBCXUk2ZFNHsOKm4mLimvCcZPemnY4sZFOJA8Oepqu1mGzxXZSXVnzc99CN7DHSo1tWDVXM9jPm7mjZ25PNX1gUYxXq0KfNHUnroSiGpIUwelcdVck7FXL1tHu61sWUe3BrpoO7uS9S3MQqc1z2oSgucHiu++hdFamexOeoNKua0vfQ67qwpJzyaY5INUtRrcAc8mlB4zSG9WG7jNJuJoaW6GJ1pvepi9QuB4BJPWk5P0oY79RSc4FJjHWpsxbBkfjSjgc02nYVtRxGRwadGSvep5XYlrua+mXJXGTW/bXO5evNE3fYxkrsvwyg4q7EfyogtTJ3TJg2BQbjb3FdEEhFeW55JyaqvOWzXXTWhUUQPJnvURUt2roiMfHbM5wBWlZ6Q8pXI/z+daqFxN6HTaT4bZgMr+P+TXVadoCR43L6d//r130odzllO7NYG3sIvmI+UYxn/69cz4o8eWmlQuzzLGqj1/+vW0nZXFCN3Y8d8U/FK51p2itJNkXQsDyf1rib26EmWdizHuTk157qOcrmsktkZr5U7vWqF+JZu3Fc9R63YXsivDaE8EdalbT9oz3qOZ7ichEsecmpltwrZPNOUtSl3Y4Rlm+XtV+1vWtMA4q6cnF6A10K+p6153yk1zd5mZsitpu4fDsNtrZtwHeugsLchBnHSs0rilLQbe2wBzxUVtHg5qEnFslN21NCFDkGrfJFXAb0GSISuKwNYhIyeTWl9kyb8pzdzKYm4qzZXjOQCTXTB6WNnL3bmoYi0W6s2eIq3OTUSjdshSvqyNAVOTTmJNc01qaOV0ebZ5oOc89Kq/Ud09BQKaxNNahFpjd3FAIJ5FK1timSBsdDilLZHJp3FdWGE5PWjFaOdlewX0sLkAUhbb2rLrcmMerEHJpQ3WrKbVxrDNLk45obUiUKpwOKcWyOalWs7laaWGbqXIJzmrba2HLQViBzyTSBvrSi3uFx4lKjr1p63W1hkmmpN6BdbFpJldck/nQUDck1TtuJxI5PwxUJX15qYWRPMlohQ2OgpVcjmtEnuNOxNDfSxn79WV1eQfeOab10f9fiacy6luPV42GOnvUgv0bjOa5alKy0G07jmuFZeDTPMyM1FnuQ9g3+tIXxzSYWQq8jPWl49ad0thXDHep4bpoh8ppqwJ66jZZy/JNEbFlOaaktUwdug1wPXmiK5kt3DIxBFCk1II6PU6LRPGk1qwEjnArv8AQvGccwQmTP410wkOUV0OtsNcjmAO7Oe5rWguFkXINavYi6Jwd3ekKYrGSHcaV700r61DRe4namkc80rDaGstMKVIX0GsOc0mPWkO40jvSZPerT0BBnNJnHWp02AAT0pCeeaNAFHTmk6Gk9R3Ez3zRmpE11DOTQTmgLaCZo6fWk9xiYo+ppsBpoPIqJLYYAcUhXNJ9h31GhfWkC56VGr1ByGsMUwj15prUaF20Y96hN7BcXbzSgc0cw7kq/rUi9KklslTmlprfQiQmMc04E0NX3I6ig980u7jis5J7oq4xnwKjaX3pvUYwyc9aieU1LetitCGSWoXlPc0nd6FWRXlfNU5jzmo5bbDWjKr565qvJnkk1i0rmpA49arTpurKeiBMz54iCaqOuDwalSckQ9CCZ2A5qlIdzfWuVx95sx5bSNTSbQuwOM896344jGvFeZi53lY9/CQ927Ek56nJqFgVHqa5onbF9yHed1SAZrRmq0F2kDrigA+tK4m0SoMkdKtxEGsZlIcfmqWJfU1i9gbJeKZtwetQmQx6rxk05kJX60mznrK8SrLlaiD89K7KTZ81Wp2kx6/MOmacsAJya1VnpbU5mi1FFgZqQtivZwzahZgnYcJPWp7ZgXxUV6XNK4O5r2lvxurSt4jWVKMosm6GX7MsfpiubunzKc5rtp3bNqPkVy3zcU4EgZrr6XZ0WsGaQtUX1KQDpnrQadrMqLE3Z60hYjvS2BpXEZsDOaTPGc0tlcXQM0mSaJaj6C7sGgnNNJsGgBz0HFKTk0tbiFHXrSg4ppikWLeYowINa1rfEAZJpqN7kWNW1vgMHNaMWoAYO6qjB7mUojzqAP8Waie/J7100oJ7k8vcZ55bqaNxPGa6VG70DRDkhLdjV2102SVhxW8KbbuZykdBpvh5nAJH+fzrqdK8NhMEpyQOf8AJrupw6mMpdDoLewt7RQ0mBs9Tj+tUNW8SW2nxnDgADqSMfzrfUzSvoeS+PPjVBp4aG2kM0vI2qen6143rXi7UfEFyZrudyM8J2Fediarm7dDojaKSIIb1yDnrU8DGV9zGphJWsKWuxLcmKJeTzVB280cdKxrK+o47akltBs6inSgHjPNZt2SuKKTCKLdSywbe3BqJNuVhsWKADtVa9YAnaeapNpjtdXMyZWYHOSTVdYGY8iuhSu7IT7l63gCckc1bW58vtRGLirsmKuyC4uTJ17UkE+Wxmkm2izQjlAXrU8VyD1NVB2dmKxJLMqpxyTWTfQmcHiqkZydlqcvqNpskPBqvbO0Ui5GDmtqd9maRndWZ0tniWEDGSaJ9NLc1clrcxbszLvYPJyAOarRA7stmueUXY15tDzrr0FIRzk80KNnYvmsOBwKa4BpW1EmMxzRlsnpRpezL3QgY+lOBwPU1b2sgvqNJx2pyt3IyabjeOjFzaC7/UUnJqOVIu3uhml2iqemxDeisIw9KcBuFEY6CWg3ofWkck96VtRp6BznJoPSm0PZBuJBoBNDSTBbARmlJyKqysTfUVXKng81MszN15przGpN7jt/OWpwdX4FFk3oFuoyQY7UznFXYFZaiA47U4kn6Uwv1AN609ZCvAJApR7MtTa1JEumHc1YivGPB6VM4J7jcuYnFyrDrThMD0rndMVrkqPke1ByamUbCsKp96MnNLqJbjuW607mravqTLewEDqaaylualPuO9hhQ5q1Z6jNZNlGI/GtIvoxxlbQ63QvG0kW3zmwB7132j+LI50Q7xg+9bwml1HJa6HT2WrpNghga0UuVk6EYrV66mZIOaCme9ZOJSY0x7RzTGHPSs9epVxp96aVpJDuMKmmlfWp6je2gwqT3pCMUvIaExjrSdKQCZ70mCTmqbsUxQSDSHmk+4uodqAfWpsxh70YzSYMAtB4NJO7EhO9BGadgGkc0c5qWAZwKTPNKwxCKMY96LOw+gjAYyT1pNoxmiwXG4GKADmko2WoX1FAHc0oBPNZtW1YDx704Gly9RXJEanbsUJWYMXOOtKW4pOxAm6mGSh+QWI2kwOeajaUGpa6lNEbP+NRPITSSuxkTsepqJm703G+iGiKQ57VXlHXIrJ6aFaldkznmoJFHcVi1qaXK8i1E6ZrGpEZVmizmqE0OM8VklYCjcKe9VRFmQVm4+7clXcjodGi2YbritoLuHSvBxT9/U9+j8GhXnj796qucdTUQd0a3uRbMZ5HNKr+vNbbl3uSK25c9KQNnk9qkT3JE4GanSTtWckUtSRSDUykkcVjIG3ccDzyaXac9c1mD8yRF9akOXHSoZlNEM8XHTJqutqWPSurD631PAx0eV3LUdmR2qUWxFdaXLqeZcds2imFa92kl7NFKIw5zUttIY2zWnKM6HTbpWAywz3FbtqY5EHTJojSUvU55poq6vEFj7VyV4dshzVciTR00G7EA5NO61rY6XuAXBzS8YqFHS5Qg4pCAaTTH5iYyKTHvVNJ7BcbyTzSkYoemg2NJz7ULx1o5WD2FzxRnFOxLFDH6UdTStqAtOCkmpVx9SaNec1ajfbVwXQhlqKdgODzU63Tnqa6IIlk6XD9zmpkZ2xzXTSWtzN2LkMbsRWlaac8uMgYrohDqZSNmw0Bpm5TP+frXU6X4YCgf3jzj/JrtpQSsYuWh01lo8dvGGkwoA5Oev60691m2sI/lZQcdSc8fnW7tuZLzOA8YfE+00qNvMnBb09f1rxvxZ8SNT15isczRQZOFDHJrlr1b+6jaC5dTiLrdJlmcsT6mqkY2nJ5NcfLe6RXxak8cm09asRy9wcUn7uwN2Wg6YFxuJos0y2WpRvLcSbe5clIVeKiii3NnPNaezurIKbaL0VsEjLcGs+9uNjYXms3TSd2HMVvtTkYpiAucnkmpur2K31I50wMDHFVnkWLnqa0pxXMNaskW4wuRSNK0hx0qpu61JslqKYywHGamjtvLwWxmsW2nZCbLKjcMCo3/dv1oXNe5V7Jk8Uu4cmpdquhrqcU9TJxb1Zg6vbBWZvWufmk2SYHUVa1ehUHc09IvwjDe3FdHHKs0Wa3fvIdTujJvrfLlieKzZ1AOFrFq6uZ36nmWcetJk9aye+pvYazc+tKW/Gk0y0hOR1pSp60eYcyDgDmmlgTxmjldyrdQPPJ5pwbHem07WJ3FXGcmlIyOTxSYr2Y1jg8ZNAbJ75qrNhbS47PPSkPXrVQTJitRM96TO4USXUp72EVsnkUvX1rNsdtNBypkd6CvsaL3F0Gk8cUDgc9Kd3awNgAOvNPDbelU7k6pCMzZyTSowz1NVzNaopN9CzvBFDR96erQJCeX2phUihKwNajcUe9NPRgAOfWnhuOtDdtBCiQ4qRZyO5o1eo7u2hKl2YxU8d4HHWlOKauWlpclSZSalRgx4Nc8l1FZkv40pOBSTstQtqMZiBSB8VLJersAkyc4pwG7qap+6CiG8jpWhp+tXFk4Ku3HvV05aWZSS2Ou0DxwVkCyuR/Wu70nxQk235xj1zXVGotjNprc6S01NJv4s1fjmBHGOe9W9RX1JMbu9MMeeaxkik0RmImkK4qLlIjK5yaaVzS3GNK00jmoavsMawppGBSaY+o3GaOlJj6iHJ6UZovoDEJwetJmhXYC5oyfWkG4bqX3osNi9eTSNSQddBtIT9aGtQ6iYpCalttjvqB55zSk5oT7iGnrQTjil1C1xARilFJhawEE0AkHmk0O4ZJ704HPU1N9LA0Lu96cHPegQvmUhl9TSsgcRpfv3ppek2JIjZiepqNnpPVDQ0moy1IY1+ajbpU8xSepDJ0qF+nNRJblLyIZAKgcVm13KRBIDUEgPXmolG2o7kMgyDVSZM9M1xyuNbmddpjOKoj5XyaIw0Y46M1dOvNpCrkmujtGZowxPavFx9Plldnr4eXukV0R3rPlOTmuektDRN3IfNwcA0gky2T2rexrcerFualXJ6VMirj1yDyf0qeM4PTk1lIZKrCpkGTWEirWJljzgU8Lz0rFslvoPQEvxU3l981nJmcmAh39qmitea7MPT95Hk46N0TrGB6VHMVUE17VLDrl1R4qi72Kjy56VEX969GEUlY25egcMPSlXOOKb2FYmhlaNgQxyPStaw1kxY3sacXJO6FycyLN5qyyx+tYMz72JPWrvzMqnGxHj0NOUcVbubrzHBc0FMDNLW1kO4nUU0jmjUQY55FIw9qNthjSvvRzVN33GJjnmgjNABkmkoVr6iQY55p2M/WlHa4bC7WBzUiJnvVq3QL9SyicVLHGT3qlF9DNliOM9e9TxQk+tbwgQ31LtvZs+MCtay0lnI4rrhTWxlKXY39N8PyPjC/Wup0rwscbiMZHp1/Wu6nA55SOnsNDSFQW6D8P61Ynv7XTUyGUsO+f/r10W7ELXc5PxJ4/t9Ohd5bkBfr/TNeQeK/ixc3++OxYrHnAkPOa561dxXKtzSEUtWed3d/LdzNLNK8jt1LHJqrPKAuSa4bvqxlNpN+cUirzuIpczDZagGycDg1YRSF6mpldNIGPLfLipLZx1zVLcpXRYZTL9KEBQ4q1N82hK0JpLxlj29BWTO+9jjmlUfNsVpYYuWIFXIYMrkGuaYJJDZIMg5FY96AshGeaqEmnqNPUZ5uE4pYJSWHJrq3QNGlbEnlhVpYmkP1rl5tdCXpsTfZgiEk1VkiMj8Vv0Iu9x/2Yx4zUyDjNDk7B0szP1aLKe5rkdQQIW9auF7XRUWk9SvZ3BjfJOK6rSroOnJ4rphJ3ZcnfZFi7AlX5eaoT2+xCx4IFZPQxj2PJT700Z6E1HLZ6mrl0FwB1FIxA4FNpk8z6CqQPencdc1nZ3Lu9xCRjnHFM4rbl2BTYuQKDg81NmF3uKMU7Ip8jHqwIAAobgcd6Vr2Ik3awdetAAHNXaz0KUugmN3ekK4HNJ7crFztOwzvSg4NTKDuO7sShsDk0jvuGKjlYXegwjvSZJHarceoXHIQ3Jp2RWkI7phfuNbnrSAhTkVPK9hXsOSQg1K1zVR7F3u7ifazQsoc81XJqHN2FJGOtAwaVnZksMelB4qYRYXAYHcUE+lOLd2TdgX7E0CQjpVNNLQFIes7KTk1YivNopSjoaKVyzFfButWFuA3fArGpTHYDIPXNN3AnOayinfUlp3uPBzQHIPWq5dXcL9Bd2TShh0qYq2w2yWI7XBWtrTtbms2XEhxnkZzW0buzZPNpqdXpHjsIwWR9p+uK7XSPFUVyoPmZ49a6ou42kldHRWmqJMgO4HPer0UyyKOabV9zOJJgHkU0x57YrCcC0yNoqYUxms0i2RlaZt5oSKT0GlMdaYy880dQVxm3mkK5qGO43nFIfrSYdwxj603nvQ2wTDNHNO2hXXUC1KM9aS8wY7d60mQetJK4mA60hpO97hfQaQc0nSk2PmEyc4oB9aUrbjA8U089qm3VgJknvxTt35UrA7MN9B5Oab93UVtQ3UhORzUW1uVEA2KXcepp9CWAc4pS2TU21GhC/PWmknFRbULCHmom5NNaAvIQmomNF+4Ja6kbMfWmFvepa0KRDIcd8k1Gx5qV5jTInphFQ43VjQhdAelQSKe9ZyWuo4voQSLVeSOuacddCkyncwkg8VnTwbcnFZSVmJq+5PpEWZxk108Q2RcV4+ZSvM9XC6QK1w+WxVGd/m4Fc1NGrID97JpVI9a6CoyHq4qVG9KiSKTLMRBGepqVE5zWDBOxOq8dKnjWs5U5WuW56FiNQvJoYYJqJUWlchSFUkGpkJbrWSpNslvQljHNTdBmu7BJxmzixK5kQSzEHrUDOSOTmvpYr3dDypKxC5z3pgGT1rTlJb6Cjg0b8dKtQvuOzY5ZCKcJT1qeXUVr7D/ADWYUqqT1qlFpDtYeI/UZp3l9zSSdyri7eM0BT1ND0WoLQY4pnNWloP1FXkcmkIyaco9QvqNxRilYJMUD1pCpzVcvUSYbSKaAKnkaHfsGKkVcinyg+5IkeamjhxVxpvczkyeOEk1ahtskcVuoNrQmUrF6GwZiBjrWrY6MXPY12UqVkYymzoNN8OPIRwfpXWaT4RO0EjAJ/z3rthS12OeU3sdbpnhlIQp2qPXP/66uzy2OkgmR1Lj+HP/ANetm9bIhJ2uzlPEvj2C2RiZ1jQf7X/168g8X/Gj53i05jK3Qtk4qKlVU1Y0jG+p59f67eaxKZ7ud27gE8Vl3V0xyM8elcEovdltlMv3qvczBjis7XFsPgj3Lk1bjgQjmnGm2yfIY9uFbIApwYYqZpplLa5HJwMiltzyCadtLhe5q28ilO1IwBJIpqNgkrFa7cIOtZ7Shc8YonFp6DhK+g2KXL+1XkulUY71DhqU30Ce8UISCM4rBuXMkpJNVyLcTukV2Zs7VrR0+zL4Zj0o52oN9SVJ7GlGqIRV5QsabuKygnpcbbsVZpzI2KdFIsfJHNdLlfQlO7Ji/mDtTccVHUXUq3MRfJJzXNazakE4/StuSy/r/Ijns7GEyeW/PPNalhf7MD1rePMbNuxu2lyGTJqK8kL5GOKmauZJcu549vHXFN/Cpa8zZaMdjHJppxnkc0tbhGVmKvHan9B0qUg5tboQ49KaAT7VfK2tyopIOc4xS7R6VVrLcGO2gd+aCppp3dmG24mSB0o6jmiyWiYlZMXr0FGSByKFa2rIaQAg801mGOalJt6mlhUQEZI5p4TPbFUtXdsd+jGv8vGKb35FFlvcBD6Yo+b0qFbqF1ewgBHJpQxBob3sw0vqDt0wKTcfTFOCvuydBRTW3ZyacdHuF0Aye1O+ZRTe9ri0uIxbqeacjstLmctinYlE3qKUke1UuyYW0uNIP4U7JAyKei0Iv3G9Rk9aQZJq4tdRqyQpzQGI7Ut3uC7DlkI5qVLx4xScb6FX1JUvGYcnrU8VymKzlHohvyLMcoPSpMDFZNWCVxuTSgMTnipXclWJEYg5zTjIwyea1jKyE7JjTcNng4rS0zxHdWLDDk49eaUZt7lKx2mg+PWdVWQsp9zXcaT4qSdR84P41vCcWTLQ6Kz1eOTGW4NX47hXOc5zVtJkpko2sM5ppjB5rGUCuZojeHvioTEc5xUWNIvQaY6jdDUvsCY3bxTGUmk0X1GsvFM20W0DYaRikxxUtjEGO9KcCpk30ATrR070MBNxJpScUK4MBmnDpRe+hTY1smkIwKlrQkY3WjcTStdFvYG5pDxSlsSmA460VK1GhO9KaqTVgvqN5PU0Z55qbgHWl7UmD3EJNGSKTY9BcjvSE1L7ghp6VGetISGEmomBNG4xhGeuajk69Km7RSIjknmmMCKbQ7jDzTGGR0qX2KW4xhjrULIetZTH1IWjz16VE0R/CsZwujRNWIJYM8Yqlc2xOeK45Ru7sdiOzhMc2a34nLR4JzXk5juj0cO7wK8657YqhcIQa5qTNWViDn1oDGukIschz1NTxjA6VEjS5ct24qdWBOBWNnzCbsWooyecVZSMntXU6TaMpTRIsZFDJntT+r8y5SVUHrFgVIqY7VpHC206mcqtyQDFIz4GCaujhnFmE53IJXHbvUZr14RsjiktbjMfN04pNuO9a8rsRbW4u31pCmRRboK4oiyBxTljxSs5Ow9bEmzkAU7pQl0Dl6kgweKeBnrVuI0IwpjNihx0C5EzY6c00NzzUtD06ijlutIxql2GxNpz0pduKdm0DfQFBzT1FW0rWFsOKgimCOocX1JTHpCSelSpDtPStIwuK5YSAnHHWrMNmSfu5rop0+hm5Gha6YZCPl962LLQ3lPK5H+feuqNKxjKTudFpvhZpMLsOO3+c11Ok+DypG8DB98/1rspwtuYTndOx1mneGki2sFwNv8Ah7/Wr8ktho8R86Vd393P/wBetG3siEu5zfiH4iw2UbBJFjUDruGf51454z+OMELvFaO08p6kHP8AWs6s1TjdbmkItvXY8x1fxfqGvOz3M77W6IpwKw3mbdz0rzvaSm7yZTa2JftJ29eKqySlnz1rXmbQ07MaQ7jpTBEx5IrNLQTkrFqORYk561LFcruy1ax9RJXRY3JIuTVeYbOe1YVYvcqzRWMhc9OP51NEjMc8ikrpaBzJItKzIvWoJL5hwOtaQeuo009xQjz43UjWGWxjiibs7oi9ncimtfK6cVWwVPXJpRkyotSGO3BxUKxhsmk229Rtk0FgWbJFaKIIU6VGjdrkMhMzGTgDFTm4YLjHNaQsNtLS4qjPJ60hBz1pbiViVZNvvT2mGOa0grssiMpcYxWZqduGjLd62qv3dDKW+hyl8mJCcVDDIVenGTVjRu5u6XMZSATW01n+4Zj6ZrSS0dzGbvseJYPIA605d235hiobizqskHel28ZxWcmgmkAVu44p31qJPsCSGkbqF961T0sJLUdwBSHPamnFrUVtQ2+tKDntUJXKsJtowe5p9mDtuKAQMUEetXpqkRy9RCAOabgnnNSmXFaaksQxyaUsc9qrTmDfUjc80wlu1S7WuK3cQhuOaXBHJoTRTauLgnk0wgk0vtArEiLzilK5OetWmkJw6iEE9KjKkdalWT1JSsGCTwafuIFW7X1KaRGST2pyk96m6SFZdQbjnNHzHmlGQ7DhIcjvU6DzBxTumTKKuP8As5xzTfJZapyQnYQg5pvPShWEhpBzSY55NXzK9h9RckUoZscEg0lJDTs0yaG5ePHtV6C9EnDnFKcVJ3L+IuIQ54OasC32r6k1m4pbENW0HRW7PIM1Yms9qe9ZX1B23KLWzg5xSiEjmi6tYV7k0Icc5wRWtp+sXNkAVkI/HrSjJGnQ6fSPHxjIWUgH612mk+L45lX96Mkev/166VVI5G+h0dprUcwB3A/Q1pw3IkI5BzWy2JsTghvQ0jR5PFZuAJ2I2hI7VC0J71hJF8yI2iI5pjIc0r9CxjLUYjwOaGtBsayUwoRWTQ2NK4pCp6079SkN6UmaFYLjjwKQDPXmktrgLSk0PuA3JHSlyWGeKW+oNDGyabnHFTqtB7geBTeetS/Ma3F5xzQMikLQKQ8UeY0GCaaTnpSQIUcdTTveh6gJkmkIxzU7MT0Dn8aMEdaHYegh5ppFLZCbI2XmmkZpIZGyVHIlL1KuRtHn1zTGjp8wDGiPam+WfSlpuUmNMRNMaI1k7NDuMaHuaYYe9ZNFrcja35ziopLPcCTXLUVtUjSLK50/a3SrUMe3g5FeVjqfuc3U7sNLSws0OfpVWeE+nNeXTuzaRnyRMhPFQ7cHNdsbErTYUSLnHeneaFHWm49w57EkN0vTNXbaXLfWqlTtJM56tbobdmu5ATVkJivRp001cyU+oYApVANdDpJaj5tBR+dOD0ciYmIZOKjdgfxquTlszOSGMtMK81rB9Tneo4IaXyzVcxAeVzQEP1q20wHpGQeadsqtLaBfUVIz+NOMealxHcAhXrTqGw9BrtgZqBzk88076hYjwSTSgcVO5TFUHPI5oMZFWkk9B6XHqhfmgRnNVDflIb1HCI5yTThGe1XJJrQHIURGpEgJ7cVaipKxLehYitCzVbh01nxxWsaehnfU0rXRWdhx79q2rLw6XP3c5rrjT2MpSOj0zwmzYO32P+c11Wl+EApQsAcHp/k10wprdnNOR1Fh4ejgwdoTHP8AnmrFzf6dpSZllVmXsDk/zq2+iEonLeIviVFaRPslSKNR1yMn9a8k8U/GiEs8do5nk9QeP51E6ns4+ZcI3dzy/wAReMtU1lm8+dgpPCKT/jXMyB925utefKWt5bmvOiW2fnBqaSMNz2pSVlchuxWmkx8opAFGCTVwTaG1ZFu3x9c0+aMEfLgmolpqHKU50de/NRIJCeDTUkxNIsxO0Qy5ps1yZmwDxVt3Wo0WbaAEZxk1MRsPNY3967BxT2JUi85TiqbWvly81rvqLyLcOAtTeYF5NTpbUfJoZuoXO7heapBmPWhWS1HypbjJM45GKigQs+Kp2S0Ki0rmvABHH05qB2eSSudPW4iWGA5yauwWBmPrWnoZPcsTWAhTpzVF1Kt0oaHFrqAUk5xUqWrTDOOlO7WqKl5DZbcxjgVn30L+WfStYtNWZkn0ZympQMSxPGKzIwQ2D61rdN2Nr3RvaOwjcEjJrqUk82Hp1FbN3i7mbSueFBwDkcU4EnvXLbU1Wwc96AxzzVOzGnoOVic5FITzSatoD2uNO7FCnnmmmhJ2YufWgse1Nbaje47GaOnrQ9di2OwAMjimE8+tJp3JWqF3elBNJNpkoaxIGaQtRLYpbgr/AKUbsnrVJsq2gZpMik3fQh9xynHWl2ZFCHLRDcEHFISQaHLVoXmOBxj3pw5NOMtdRXuDCmN05pz1s0N7jN4FG7PWm99QVgGPzoLc9eKgbEHWn59al7k3d7jWbninLOyYIqrh6k8d9/eFWFmRxnNaWVrlpC7FY56014cc4qLu4miu4596aBxWuhDYh60c1FwsrCkmjeQc1bvcIvqW7S7aLmtWDVV25cVb13/r8S2ky3DqcDHPQ1I2px9C1Y+zsFnsMa9hJ4ao2u0PAqZU77CULMfHcLjgU8zjnmsXCyG13IWnI5Bq1ZazdWrfJIw5z1peo07Kx1Gj+P5YCBMSQD2712+j+OorjH7wfTNdMZ2J5bK51Nh4jinA+cVrw6gjgYYe9btvYySLIlR+9OKBhnrUSiGqI2gyKheD2rFxsaRZC8BDVG0ZzzUPY0vcYyYqMpk1NrgMKkmmlDQ4jTGFeaaRjrUNalXuIBzmgkimJdwLYpN2eaB2YtAxml1GriMaYSal9w6iGjPGKnoUHejNG4WAc0Go1vYXUTNITQhiqATk0p9SaWomNzzQWzVaMLahmmknrUFBkkUhp20sJoa2fWm0rKwDT1pkg71LDdkZphUUpDE28cmkK8UW01HcYRTSv41lbUaGlSaQx96mUexo7DTFS+R61jKN9ilIb9mzyactt7VxYmjzp2N6c+Vg1scdKqXMJGeMV51PDqOrX9fcbyqoxr5WBPWsq4mZSRzVqk10MvbLYqvdsDnOKgkvJDn5jiumEVbVGc8QkOtryQuOeK6TSnMm3IqKy7f1+JyzqXZ0tqp2jNWC3GK78K+eBtTldDTnOSaTdzmul67GtxckjOaRWPJ61NhN6CZLHk0bSafLrqQ2O2kDmmlTmi1kYMcq5p496cfMlikUqrzVJX1BbC7ad05rRpaIEKF7mnEUv7oMax9eaYzYGOlKS7ARMxPWmHIqWuZ6FAqlvWn+Xx6VaVtGDHLFkc04RY5xVWs9CLj1jpyxVaj1QXHLD61IlqT2rSnAhssRWDMcEVeg0hmH3a3UFsQ5Gra6CzkfLW3Y+GWYL8vX0rphBsxnNJnR6Z4Q34yn5/8A666fTfCyJglBx/n1rqjDuc7nzM6C20aO2QGTaB7/AP66Zd+INO0tSFZXYeh4/nVXcgWhx3iX4opaxOTcpCo7BscfnXkXin46RSO0di5mfOM7uP51lUqKCstzWEHL3mec654t1DW2LXVy5DfwqSBj+tZkUwXrkk1ytSb5mxtvRInCK/zNiqd06A4zzUOnpcnW90VlOG5NSyy7I+D1p67MblcrxKS2TT9hzndRfXQfN3LEbrGOCSasLOoXJNZO8nqXFOxDJN5pwopQNg6ZNJQV7InlS3IZo2PJ5otId7c9q1d9hX6I01uY7aPHBNUbm98xsioUUmpDStqaOmy7wAeKvtZLINx+tbP4bIhsoXRW3J5FUnny3WuZt7G0WkrDjGHTJqlP+7JxzVct1oZ3uyszFqtW6KuGI5NE78uhSV2aNvHvHFK0So3PFZJbNkvXQc7qo4qzZXBUYNbU7W1CMWtyzNMJF68CqUqgnirnsKwqQhu9Xok2x+orJO7sTsrkMu0g1TnQMCDWvKtgT3ZzGvwCMGuaRP3vNXTLh3L9pNtcAV1ulvmIBuvvXWrvQme54auS1SBiDzWDS5jTYcTkUmMHrU2s7DcrC4I/GlJOcUpNXFJ2EOaQAhs1ceUd0KRwSTikHTg0O1rFrYUHFPyAKehL7jS2OppvU1TS3RSloA45zQWI5JzS5k3dEp66iMxYdaQdOTSly2C9tRQo65NKoqY7sfOwbAHWm5x3oUVuClcUEZzmlD89aqK0E2ITnmjORg1Fk2Jt6Aq981IgC9+apyXNZFNiM55B6VCTweTV2iloTcbkHtQAckmp0vqU5WQ7AAo5qVqhN6B2pNxq7RvcG+gZNJyetTpcTl2FC+hNODEd6akrWL5iaC4KnFXi5aPrzVaMUndFKTIc96Qjmqsr2IT1DilAyaOVDdxVQsak+zFuaNLiUrIesBX6VIFPSk9xX0G5YcjNNMrjqTmnFlcyGmR+uTQLqUH71XFrqVzrcmW/kA61LFeuec1LUWm2S3fQsJcAjJap45h1yKxlBdRu9h/md881Pb300DZSRgfY1mn5DjLodJo/jK7s9oZiw967TQ/HscoAd8N3zx/WtPaK+4pRutDrdO8VxTYHmDH161t2+sRPj5s10pxbsZWaLsd0knQ9alyD0pOK2C7GtEDUT2/esJQGn3IZIPUVEYDWfKaJ9CJo+elMMZzQ+xdhjJzzTWjyKmSW4xjIaYUIHSot0H0GFe5NJyO1CKTE3Gk3GhpbjQuc0wnFRpfULi+9JzmkNMU0dajqK4o55ox61TsFxp6UnIpK2oIBmhmIFD10GNGfWg1LdnYLgDxQeaLAxRxTSQeaS12E+4nU0w4zQkWhGHFMOadlYnYYwOabjk0nFMaY3p1oYZFS7XHcbtzSbBUuwX1EKelKE9aizHe4piGaUKAM5rJU7jTBVDGpBGM1Elry2NRrjHeqk8ZbjFL2ELaopSM26sd2eKybnSixOBWMqEen9fgYVO5Ql0Zsniqx0dycBa5HCUf6/wCAcjqO5e0/w6zMDtJrqNO0byFBKUnRctX/AF+Bm6mpppDsGKCK7aMeXRHdRb3EKcdaQR10W6m/MBQ0oipu1kwbF2YOaXbzmi1yZMXZ3prLTlG5ixAOM5xSgfjTsthsUfrTgCe9OySuLzHYNOC+tOybJuKRjnNMLVVtRoYz47UzBaiyTuPzE8s/WpFhJ60nDqhskWH2pwhI5NCj/MJscIj1p6w9K2tpoQ9iRYN3Y1LHYlj9auMWiea2hbh0olhkVpW2hsWBI/T/AOvXRCDIbZsWXh5mwAMA/wCfWt2w8KtIAQmfpXVCl1MHUR0Gn+FQvJH0/wA5rfsPDoi25UADuP8A9ddMYpGDdzVENrZDdNKo46Z/+vWdqXjay09CLcDIGOf/ANdJu5SSRwfij4tQW295bvAA+6G/pmvJvEvxpubqVorFWXP8Z4/HrWVWry6RLjFJczOD1fX77VSz3V1JIG7ZwKyVAznGBXFu79Sm3IkZwq9zUaSMW61V3sxXsTzTOY9uSM1V5GSeSKUp3GpdBFcqcnrTuZD81F0V0LCphOlM25PWpRkg3E/KBT1jdjg1LaRbnbYmWIKKlVd3WlbW4o36iXDKiEYrP+0FCcda2eupVrDVMkuSxNTRwAHJJNJyE5l2CcQnp+NTS62yDAzg0oS13LjFWuyhcXfmnPNRLIepNZzV3cLE0crH3qG4jJySc1f2dCW9SqFO6rlvGzsMjrQtVYHLQ2beMRRZ71BKpJPFKcbImGhHtzirkUZwKxSdtByb2ROEIHNM8sljit7q1mSpe7qT28OTuPNTynaO2Kzt71wTVyhM+cnNQYJBNarbUp2Rg69ESuT1NcjdkwsTmqSsxqV9BtlMWmBBIrr9IvBtAzk4ruTTXmEt7I8bG7dnIwKeCRXF7tytNB2D1pQetVy9UU0hQaQHJ680rJsiUbi85pSaemyEl0EPzjFIq44pPZo06WY/imvnPFKKs9RXVwPI6UhzinfXQpNdQUZHvR9apLXUh9hrDNGCPShpD0sLjjOTRuNK66EMQ5PSjaQKi+pUbdRpGaFyOtaX0sEtdBc4PNHLc5pNLcLaaDs7aNxptK/MFtNQY5HWmHikmGnQX7o5pME80Rs2KVmhGDUmTTTSugVhwbimnINToK1xyruFIwwfX61UV1EJk+tGcCmo32LRLAu5+BWiMhOe1W0low6MpudzZzRjIpLa4m7KwgGTip40z1przFvuTIgzVmOMHrihIGhzx54AqPyOal2DbQR4fQ81XeM5xTXmTZdRojY9eaa8ZHaq0bG7ETAgZoEjrzTVr2DtcdHM2ct0qeO6OetOUTRtdCdbwmrUExY9RUuKvYltW0NCOQBRmrNtPtbIcjFZclmNaK5q2viK4sjlZWP1Nben/EkwsqyyfWtlCMUD95XR2Oj+PILhQRMv5101h4kil6uPempX2MlfU1IdTjfGHHNW1nV/T86tpMFrqP2h6abcHpWLhuUnYhe1Oaha155rNw6lKepC9uQc4qJoTmpktC0xhiIpjR461PKUhnl89KYUOeRUtDuMaOmFDilYqLGkECm7e5FQ7DTDPrS1DvcdhB1penNDWoNiqcUE5GaBDTx1pCaNGwF6HmhhuqWD7iYC00nFDXUYA80Gk9B2EOc80YpLyDQD9aafakguNzzk0hGarzBa6EbAZzTMdzU3H00Ex7UhHvR1BDec0uMnJpPQNtRCuKVc+tNvQaYpx70gz6VO4JMcOO9BJNYpK92aJiNkioym481XLzPQpvsRtbZ6io2sgajkvqZt6WGHTFb0po0lC2SKSoqXT+vuOSa1LttYpFjgE1bICimqUYsz5NdCNmqM4zUJO90dsHbQcvrQwq3Y0vqBFAzU2sN2YAZPNKoyeauO1yWxSKjYc02QnoIBk805Qe9PcPMeEJ5pwX1otfQTdxwSgnmmkkJDWJJpNpNK9ndlaAIjnmpBAc1VuYljxBke9PEPtVRj0Ym+o5YfapFtyT0rVRJciZLIsemc1ai0t2xha3jTViHLoX7bRWY4wOf8+talr4fbIypBFbQp3M5SsbVl4YZ2HyHj1rds/CoU/T/PrXTClymMpt7G9p/htAo+XP8AX9a1odLht0JkYIB3z/8AXroVkY7u4251zTtOTG5WYeh/+vXN618RUt0I85EUejDP86m6Wsi7X0R5v4o+Ndpab1W5aaTsqnP9a80174r6tqwYQyNBGff5q5KtXmWhso8quzl2v5byTzJ5Xdv9ps1XuD5jkg1g720IbbInBI6moAXD89KmLH0LBwBz1NMUhCTQt7k6bjZrjcOtVHuzu2Cmo3KSJoAzDJPFWFKqck1PMuhS7EjTAAUxWyckmlsJR1LUEG5skZq15WwdOayl5juRvwOaZGSa0S92499AuI2ZOtU47Rt3PSlzJAnbQtFFiXnrUfngGlBaXIUSOSYk8UzaZDnFNxsW2tCdbcAZIpBDuPNEtuYL6FiCEUXEalalS0ZMl1KscOZPmArUtbYcMQKuKtZg9i5gAVRuHO7rWtRJrUUdFqRxfNIDWvbbduahR0sU1oMmO48UIrAURskCtsSRtt5JqOefcD6UJ3VxMpSyGkWcAckVotdxuKZk60dy5xnHSuH1UsJSaatcUFqVoJtp9K6DRbn96nPcV10mtUypeR5mMjgk0/JIrlklctS1sOXOfm4pxGe/FU3ZaBqNA9+KMgECo6jb10JAcCjdSSvqyW9hMY5pDmqVhp6jiTQckc0ebFa+whBxRjHJqtBoOfXAoJxQ46itrcaM5zS4OKUkVdPUaScelKPek7LQlgOScUuCO9VZW1K02Ar8uaQjHU01FatkxWowjNPVNpoa0K2FYZ561GcipasK6FB5oB3H0pcr3Y0rbiEknvilHHBq1GyuIOcZpuDRyk3QoUmk6Hmk0VFguc9aGPODS62BPUb04pyoSat7Bpcv2VsfvEVLcyBeKVrky3KhagHuarlaGxQc80GUhcA0WsgiEUzZ5NTi8K9TVOVmU7bDhfDruNIL/HOadmHKt2L/AGhuphvQx5qWruwrokWUPyDTWbd3ptJEtBsG3pUbR88Cku4W6iBQByKciD0pu4mx2w1JFI6HrimmF1cn+2vtxSJqMq9DTurlvltZj31R2XBqI3TNzmm31QXsXtP1eeA5EhGPeum03xveW2Cz7setGkuuoNKR1ej/ABNj4Er7Sem6uw0rxzb3IX98uP8AeyKG7Gcou50Nn4jjlwN6/nWrBqUbr94ZqnZkN2LCTrJ0Oak2qw7VnOHYNURtbhuRUT2nGazlA05u5A9oc1FJaEHpWfKVGRE9uQaiaE5yRS5ddS0yMxNmmNFjkiol5FJjTFmo2ix2qLdBpjTHgZpNtHK7FrUQijHrUeRLYEflSEGlbW41qIwzSAYoj1BPuPVKcF9adk9Qb6jHT2qNhionsPcZS59alvUYnejdTDqITxSEmk0OwZ5pCeKTTFqiN6jORSQw5ppNWorVksCOetBXual2e5VwIzSYxS2VguOoHvUSeggNJuI60WjsWmL9elGeaLLoJscRkUoQY6UpLoLWwmwUpAprclgKR+e9TKV0ToNwaCves0mmWtEApQM1S8x3HFaXbQ9kCYm3Bo6GtFFIL3EIPXNN2nNRuwFCnNPC4FVy6gOVTSkVdrO4IBnpSiM55pNa3QnKw4Q8VIsOOvWq5NNUJskWD2qQW571fIyeYelqc8Cpo7BmPINaqDb1Ici3BpLHPHX1q/b6C7EHbW0KavdkynfU07bw6WxlSa17PwyxOdn6/wD166IwMZVO5tWnhpV+8AD/AJ962LTQFVQNuQe/+TW8adjJy6mpBpUMIzIyr+OP60smpafYjBkVmXpg1qT6mXqHjyK3DBHVAB3P/wBeuI8S/Fq1swfNvVyOcb+fyzWcpqO5UVfVHmGv/G+a4LCxUsD0kZuP/r15/rfjPV9bdhPdP5bfwg4FcFSo5vR6HRFqJQhR5ACSfrUr4TiovfQzk9RqPg8nrTjIiZJq+X3dCVe+g2OVZD7UyT5WyKlWTKa7kMjk81CXJ7022tCumoO3HqabHbNLJn1pKW7CxeW1aJP61C3Ws43uTFkqZbqeKeoCH71J2LvroX7WZUXpT3l79aUooaK8lwQee9WIV4DmiSSjoPrcdkM2Cc0rRZXIFYtdgfcpTowOTUWN1bp3ViO4NEeuOKdChY+1KLuSmW9uAOKRouKVRO2hV9BI5PKzk1HNOCcg5oa925drkatg7jWna3BKjsK0hK+5MopoluLnavWs2SQs3HNW3oLdWY9ZSmB0q7BdYFZRb5hrVFlDvxk9at7cJ0pLfUOpVdstiopGHStEtG0KUXuU7k4BrOe52H1px2sO+hU1C5Doa5LV3BbiqtqioaGZHIS3TitXTrhopUbPANdlKy3FI4okA9BzSqBiuaTvqOTQoAJJpSuRxRdl3SBVYdcU1kO7J60rq4JpMXOen50YJ61TSiS7Du1AqE9GDY8LnqKGGBk1aHdIb1oII7U3uPmVgIUdaBk9elTzaXYJphjFNboaTldXBNXsNHHWgqxoum7ho2A+lLuPeqT7hdCg46nmkK04yBy00HBQOeKcMZ5oTIuNf2qPaT1qE76sqNgABJGKAMHPatL6D59NQLZ+lDZODQrWE2AJ6k05eR0oe4ptW0HOMCoz70Saa0BNco7PtTXX86zsCasAUVZtIQzc81auwepfkKwx8CqMshc81cVZCRHjvR160RBscAKic4Yindj5u4inaaGbOab1Vxp6XDdxQB75ovYW2g4e1IwOaSaV2yJaMAxGecUiuynrVRlfRlOehZiuOxqypB64pSSWoPWw1lHXrRgelK4rCrwc0MQaLaitcYc5pwHHIoaHe4hj70mxsY6UKQlLoxCWTmpUu5F4zVxfct2QovXVs7uat2ut3UMm4TMMe9Nu+5SmmjqNG+Id7bsqySFl+tdtpPxNhYqJHKfjU2ZDjroddpfjeCcfLKpz71u2fiKN+rL+dUnfYxs9malvqUcmMMOferKTq/O4U3Z6BqSfK3BpGgB7Vm4roUnpYia0B69qiezLAnFRKPcamQvZknpUD2hNZSjbY05iI2xB6VG1vknilazKuMMBHWo3i9qhxbHzEZhOeaRkPpS5dCrpsaUPem7D70rFXDYRRg1HLqLRi5I7UZptWG7ATkc1G656VjLzBX3GbM+1JsOM0hpiYNBWnqmO41hzxSMO9F7DuJQRih6qwXGNzzTCM1KQXEpGyTTAbtyaU5zzTG2hGz6mj3NElcHsOHXNBHNZrXclDcGkPJxV2SHcBxTupqLagOU4PNKTzTlrsK4ZNHNJ6ISdg2kdDScnrSa0DcVQcc0baSB66Bspdu2ptoFhR0oPNW9rMa3sG3vRtzUq8hsNhzmnLH61fKK4uyl2H0q7BuPWEmn/AGck9KOVvQXNYelqakW1PXvVKDsTzIlW2YnGKmjsmY52nmtlDQiTJ49NZwKtw6O7EZXFaQpkSfU0LfQmcj5O9adr4bYt9w1tGDRk5qxrWfhrbgsvHTr/APXrWtfDwG3K5HT/ADzW0aWqbM5T0satvoSqQAAAR6/z5q+unwwgb2Qfj/8AXrdJIzb6jZtUs7QHBUke/H86yr7xnHCp2OqADpn/AOvRJ9SlfoclrnxPtLRGMt2P++v/AK9ee678bIyWW1Uyn8qwqV+XRbmqp395nCax8SNZ1NzuuPLX+6hIrn5r6a7JaV3Zj3Zia4223eTKm10KUuScdxTcEEZqHdhzI0IHCx5Paq893vbgUWvqCityPzHJ4pHjduSa057Wigukia2SQkZ6VYlj+XAHNTJpslNNlSSFu9MaIIvqaTlfYpvQhdHzyMVq6UY0U7hjA6+tFrrRjv0RYu7iMrgGs8wGQ57ZoirMS0RMsAUVAUJYnsKHpcaZPCzsOmBU0RJbk5qW1ewXSuWVtw/zEZNJI2eAaiS5dxRGxKA2c1oRFWGMVHXUp7Fe8iBBAqnHBg5IrWasvdM762RNJDwM0sUeSMCpastAW5JIuzk0g+YZq4u6Kb0K8wwagMO7mob1BMekRbg1ciBiXA5q423Jb1BgZM5FJ5CouaqWiKUtSNYfMbpVhYdpFZxXcL66F23TjcRnFTTzgLwabmuYbXUqlgDmoHmy2BVpqw29UVLqbqKxrqcKDVWtqFrGTe6l8pVelYF5KZmq7JvQ00iiFU5qxGTkfWtouzsS2f/Z\"\n", + " }\n", + " },\n", + " \"params\": {\n", + " \"score_threshold\": \"0.8\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "r5pn1woLyabF" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ROWzLAszyabF" + }, + "outputs": [], + "source": [ + "request = clients[\"prediction\"].predict(request=prediction_request)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "K91imH7ByabF" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "TPJAlYlKyabG" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aeeb6b7e4232" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"payload\": [\n", + " {\n", + " \"annotationSpecId\": \"3163553338344079360\",\n", + " \"imageObjectDetection\": {\n", + " \"boundingBox\": {\n", + " \"normalizedVertices\": [\n", + " {\n", + " \"x\": 0.019032001,\n", + " \"y\": 0.047988813\n", + " },\n", + " {\n", + " \"x\": 0.5898183,\n", + " \"y\": 0.7881106\n", + " }\n", + " ]\n", + " },\n", + " \"score\": 0.9619299\n", + " },\n", + " \"displayName\": \"Baked Goods\"\n", + " },\n", + " {\n", + " \"annotationSpecId\": \"6622317852164620288\",\n", + " \"imageObjectDetection\": {\n", + " \"boundingBox\": {\n", + " \"normalizedVertices\": [\n", + " {\n", + " \"x\": 0.55052197,\n", + " \"y\": 0.5830328\n", + " },\n", + " {\n", + " \"x\": 0.63364756,\n", + " \"y\": 0.7483183\n", + " }\n", + " ]\n", + " },\n", + " \"score\": 0.8705111\n", + " },\n", + " \"displayName\": \"Tomato\"\n", + " }\n", + " ]\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "j14fFDBnxXii" + }, + "source": [ + "### [projects.locations.models.undeploy](https://cloud.google.com/automl/docs/reference/rest/v1/projects.locations.models/undeploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wrhN_FT4xXii" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "kR-o0CO7xXii" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].undeploy_model(name=model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "JF1qgjOfxXij" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "-VbCkwYyxXij" + }, + "outputs": [], + "source": [ + "result = request.result()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "SpuhCYjSxXij" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ot9hRqTVxXij" + }, + "outputs": [], + "source": [ + "print(MessageToJson(result))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "DYzaoP0CEV7j" + }, + "source": [ + "*Example output*:\r\n", + "```\r\n", + "{}\r\n", + "```" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bQ-VVaSxJjkd" + }, + "source": [ + "# Cleaning up\r\n", + "\r\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\r\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\r\n", + "\r\n", + "Otherwise, you can delete the individual resources you created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "zFHJL5aOyabG" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_model = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the AutoML fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset:\n", + " clients[\"automl\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the AutoML fully qualified identifier for the model\n", + "try:\n", + " if delete_model:\n", + " clients[\"automl\"].delete_model(name=model_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r gs://$BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "UJ5 legacy AutoML Vision Images Object Detection.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/migration/UJ5 unified AutoML Vision Video Classification.ipynb b/notebooks/community/migration/UJ5 unified AutoML Vision Video Classification.ipynb new file mode 100644 index 000000000..726192755 --- /dev/null +++ b/notebooks/community/migration/UJ5 unified AutoML Vision Video Classification.ipynb @@ -0,0 +1,2480 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title:migration,new" + }, + "source": [ + "# Vertex AI AutoML Vision Image Object Detection\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest (preview) version of Vertex SDK.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "p8ShaoMNV2Et" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-aiplatform --user" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the Google *cloud-storage* library as well.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "UXzBtyODV2Eu" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-storage" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the Kernel\n", + "\n", + "Once you've installed the Vertex SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "5mnVqHBxV2Ex" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "if not os.getenv(\"AUTORUN\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU run-time\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your GCP project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a GCP project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebooks.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex AI. We recommend when possible, to choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You cannot use a Multi-Regional Storage bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see [Region support for Vertex AI services](https://cloud.google.com/vertex-ai/docs/general/locations)\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "l0Jf8RAOV2FN" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "P0HtKsWbV2FQ" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your GCP account\n", + "\n", + "**If you are using Google Cloud Notebooks**, your environment is already\n", + "authenticated. Skip this step.\n", + "\n", + "*Note: If you are on an Vertex notebook and run the cell, the cell knows to skip executing the authentication steps.*\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "o_xDBWYNV2FS" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your Google Cloud account. This provides access\n", + "# to your Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Vertex, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this tutorial in a notebook locally, replace the string\n", + " # below with the path to your service account key and run this cell to\n", + " # authenticate your Google Cloud account.\n", + " else:\n", + " %env GOOGLE_APPLICATION_CREDENTIALS your_path_to_credentials.json\n", + "\n", + " # Log in to your account on Google Cloud\n", + " ! gcloud auth login" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n", + " BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "qh1y6iuPV2FV" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "kGcs-GqmV2Fe" + }, + "outputs": [], + "source": [ + "! gsutil ls -al gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip" + }, + "source": [ + "#### Import Vertex SDK\n", + "\n", + "Import the Vertex SDK into our Python environment.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Frc6SDm-V2Fh" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex AI constants\n", + "\n", + "Setup up the following constants for Vertex AI:\n", + "\n", + "- `API_ENDPOINT`: The Vertex AI API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex AI location root path for dataset, model and endpoint resources.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "kso4G74bV2Fk" + }, + "outputs": [], + "source": [ + "# API Endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex AI location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "automl_constants:automl" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Next, setup constants unique to AutoML image classification datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the managed dataset service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the managed dataset service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the managed pipelines service the task (e.g., classification) to train the model for.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "automl_constants:automl,icn" + }, + "outputs": [], + "source": [ + "# Image Dataset type\n", + "IMAGE_SCHEMA = \"google-cloud-aiplatform/schema/dataset/metadata/image_1.0.0.yaml\"\n", + "# Image Labeling type\n", + "IMPORT_SCHEMA_IMAGE_OBJECT_DETECTION_BOX = \"gs://google-cloud-aiplatform/schema/dataset/ioformat/image_bounding_box_io_format_1.0.0.yaml\"\n", + "# Image Training task\n", + "TRAINING_IMAGE_OBJECT_DETECTION_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_image_object_detection_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients" + }, + "source": [ + "## Clients\n", + "\n", + "The Vertex SDK works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the server (Vertex).\n", + "\n", + "You will use several clients in this tutorial, so set them all up upfront.\n", + "\n", + "- Dataset Service for managed datasets.\n", + "- Model Service for managed models.\n", + "- Pipeline Service for training.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving. *Note*: Prediction has a different service endpoint.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "zo8iW2ivV2Fv" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "clients[\"job\"] = create_job_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:flowers,csv,icn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://cloud-ml-data/img/openimage/csv/salads_ml_use.csv\"" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "qIn8ZjAES2vE" + }, + "outputs": [], + "source": [ + "! gsutil cat $IMPORT_FILE | head -n 10" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "5HT7aDx5S2vF" + }, + "source": [ + "*Example output*:\n", + "```\n", + "TEST,gs://cloud-ml-data/img/openimage/103/279324025_3e74a32a84_o.jpg,Baked Goods,0.005743,0.084985,,,0.567511,0.735736,,\n", + "TEST,gs://cloud-ml-data/img/openimage/103/279324025_3e74a32a84_o.jpg,Salad,0.402759,0.310473,,,1.000000,0.982695,,\n", + "TEST,gs://cloud-ml-data/img/openimage/1064/3167707458_7b2eebed9e_o.jpg,Cheese,0.000000,0.000000,,,0.054865,0.480665,,\n", + "TEST,gs://cloud-ml-data/img/openimage/1064/3167707458_7b2eebed9e_o.jpg,Cheese,0.041131,0.401678,,,0.318230,0.785916,,\n", + "TEST,gs://cloud-ml-data/img/openimage/1064/3167707458_7b2eebed9e_o.jpg,Cheese,0.116263,0.065161,,,0.451528,0.286489,,\n", + "TEST,gs://cloud-ml-data/img/openimage/1064/3167707458_7b2eebed9e_o.jpg,Cheese,0.557359,0.411551,,,0.988760,0.731613,,\n", + "TEST,gs://cloud-ml-data/img/openimage/1064/3167707458_7b2eebed9e_o.jpg,Cheese,0.562206,0.059401,,,0.876467,0.260982,,\n", + "TEST,gs://cloud-ml-data/img/openimage/1064/3167707458_7b2eebed9e_o.jpg,Cheese,0.567861,0.000161,,,0.699543,0.077502,,\n", + "TEST,gs://cloud-ml-data/img/openimage/1064/3167707458_7b2eebed9e_o.jpg,Cheese,0.916052,0.085569,,,1.000000,0.348036,,\n", + "TEST,gs://cloud-ml-data/img/openimage/1064/3167707458_7b2eebed9e_o.jpg,Salad,0.000000,0.000000,,,1.000000,1.000000,,\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_a_dataset:migration" + }, + "source": [ + "## Create a dataset\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_create:migration,new" + }, + "source": [ + "### [projects.locations.datasets.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.datasets/create)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "request:migration" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "datasets_create:migration,new,request" + }, + "outputs": [], + "source": [ + "DATA_SCHEMA = IMAGE_SCHEMA\n", + "\n", + "dataset = {\n", + " \"display_name\": \"salads_\" + TIMESTAMP,\n", + " \"metadata_schema_uri\": \"gs://\" + DATA_SCHEMA,\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateDatasetRequest(\n", + " parent=PARENT,\n", + " dataset=dataset,\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0zipRM1fS2vI" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"dataset\": {\n", + " \"displayName\": \"salads_20210226015226\",\n", + " \"metadataSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/metadata/image_1.0.0.yaml\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "call:migration" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "datasets_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"dataset\"].create_dataset(\n", + " parent=PARENT,\n", + " dataset=dataset,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "response:migration" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,response" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "MAwq6XTQS2vK" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/datasets/8577474926234042368\",\n", + " \"displayName\": \"salads_20210226015226\",\n", + " \"metadataSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/metadata/image_1.0.0.yaml\",\n", + " \"labels\": {\n", + " \"aiplatform.googleapis.com/dataset_metadata_schema\": \"IMAGE\"\n", + " },\n", + " \"metadata\": {\n", + " \"dataItemSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/dataitem/image_1.0.0.yaml\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new" + }, + "source": [ + "### [projects.locations.datasets.import](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.datasets/import)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "feizDpanV2GG" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "outputs": [], + "source": [ + "LABEL_SCHEMA = IMPORT_SCHEMA_IMAGE_OBJECT_DETECTION_BOX\n", + "\n", + "import_config = {\n", + " \"gcs_source\": {\n", + " \"uris\": [IMPORT_FILE],\n", + " },\n", + " \"import_schema_uri\": LABEL_SCHEMA,\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.ImportDataRequest(\n", + " name=dataset_short_id,\n", + " import_configs=[import_config],\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "8YwszqJ_V2GH" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"8577474926234042368\",\n", + " \"importConfigs\": [\n", + " {\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://cloud-ml-data/img/openimage/csv/salads_ml_use.csv\"\n", + " ]\n", + " },\n", + " \"importSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/ioformat/image_bounding_box_io_format_1.0.0.yaml\"\n", + " }\n", + " ]\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "p6gKpzGlV2GH" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "datasets_import:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"dataset\"].import_data(\n", + " name=dataset_id,\n", + " import_configs=[import_config],\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ZY3kU6P6V2GS" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "O3aD3ug2V2GX" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_a_model:migration" + }, + "source": [ + "## Train a model\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "trainingpipelines_create:migration,new" + }, + "source": [ + "### [projects.locations.trainingPipelines.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.trainingPipelines/create)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "VUKADaK5V2Gc" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_create:migration,new,request,icn" + }, + "outputs": [], + "source": [ + "TRAINING_SCHEMA = TRAINING_IMAGE_OBJECT_DETECTION_SCHEMA\n", + "\n", + "task = Value(\n", + " struct_value=Struct(\n", + " fields={\n", + " \"budget_milli_node_hours\": Value(number_value=20000),\n", + " \"disable_early_stopping\": Value(bool_value=False),\n", + " },\n", + " )\n", + ")\n", + "\n", + "training_pipeline = {\n", + " \"display_name\": \"salads_\" + TIMESTAMP,\n", + " \"input_data_config\": {\n", + " \"dataset_id\": dataset_short_id,\n", + " },\n", + " \"model_to_upload\": {\n", + " \"display_name\": \"salads_\" + TIMESTAMP,\n", + " },\n", + " \"training_task_definition\": TRAINING_SCHEMA,\n", + " \"training_task_inputs\": task,\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateTrainingPipelineRequest(\n", + " parent=PARENT,\n", + " training_pipeline=training_pipeline,\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wXYRJLf4V2Gk" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"trainingPipeline\": {\n", + " \"displayName\": \"salads_20210226015226\",\n", + " \"inputDataConfig\": {\n", + " \"datasetId\": \"8577474926234042368\"\n", + " },\n", + " \"trainingTaskDefinition\": \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_image_object_detection_1.0.0.yaml\",\n", + " \"trainingTaskInputs\": {\n", + " \"budget_milli_node_hours\": 20000.0,\n", + " \"disable_early_stopping\": false\n", + " },\n", + " \"modelToUpload\": {\n", + " \"displayName\": \"salads_20210226015226\"\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "9wQgjU9AV2Gl" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT,\n", + " training_pipeline=training_pipeline,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ogw8qBSkV2G0" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,request" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "trainingpipelines_create:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/trainingPipelines/2049683188220952576\",\n", + " \"displayName\": \"salads_20210226015226\",\n", + " \"inputDataConfig\": {\n", + " \"datasetId\": \"8577474926234042368\"\n", + " },\n", + " \"trainingTaskDefinition\": \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_image_object_detection_1.0.0.yaml\",\n", + " \"trainingTaskInputs\": {\n", + " \"budgetMilliNodeHours\": \"20000\"\n", + " },\n", + " \"modelToUpload\": {\n", + " \"displayName\": \"salads_20210226015226\"\n", + " },\n", + " \"state\": \"PIPELINE_STATE_PENDING\",\n", + " \"createTime\": \"2021-02-26T02:12:41.612146Z\",\n", + " \"updateTime\": \"2021-02-26T02:12:41.612146Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "training_pipeline_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the training pipeline\n", + "training_pipeline_id = request.name\n", + "# The short numeric ID for the training pipeline\n", + "training_pipeline_short_id = training_pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(training_pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "trainingpipelines_get:migration,new" + }, + "source": [ + "### [projects.locations.trainingPipelines.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.trainingPipelines/get)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "BfPthv3CV2G4" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_get:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"pipeline\"].get_training_pipeline(\n", + " name=training_pipeline_id,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "PoZIyA46V2G5" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "3OVVM56XV2G5" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "trainingpipelines_get:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/trainingPipelines/2049683188220952576\",\n", + " \"displayName\": \"salads_20210226015226\",\n", + " \"inputDataConfig\": {\n", + " \"datasetId\": \"8577474926234042368\"\n", + " },\n", + " \"trainingTaskDefinition\": \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_image_object_detection_1.0.0.yaml\",\n", + " \"trainingTaskInputs\": {\n", + " \"budgetMilliNodeHours\": \"20000\"\n", + " },\n", + " \"modelToUpload\": {\n", + " \"displayName\": \"salads_20210226015226\"\n", + " },\n", + " \"state\": \"PIPELINE_STATE_PENDING\",\n", + " \"createTime\": \"2021-02-26T02:12:41.612146Z\",\n", + " \"updateTime\": \"2021-02-26T02:12:41.612146Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=training_pipeline_id)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_name = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " break\n", + " else:\n", + " model_id = response.model_to_upload.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(20)\n", + "\n", + "print(model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "evaluate_the_model:migration" + }, + "source": [ + "## Evaluate the model\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "models_evaluations_list:migration,new" + }, + "source": [ + "### [projects.locations.models.evaluations.list](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.models.evaluations/list)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "X47P3WEMV2HC" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "models_evaluations_list:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"model\"].list_model_evaluations(parent=model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "G7i7OmGRV2HD" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "AxEY7WFHj2YC" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "model_evaluations = [json.loads(MessageToJson(me.__dict__[\"_pb\"])) for me in request]\n", + "# The evaluation slice\n", + "evaluation_slice = request.model_evaluations[0].name\n", + "\n", + "print(json.dumps(model_evaluations, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "eW5ospNaS2vV" + }, + "source": [ + "*Example output*:\n", + "```\n", + "\n", + "[\n", + " {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/770273865954754560/evaluations/7557961565471768576\",\n", + " \"metricsSchemaUri\": \"gs://google-cloud-aiplatform/schema/modelevaluation/image_object_detection_metrics_1.0.0.yaml\",\n", + " \"metrics\": {\n", + " \"boundingBoxMetrics\": [\n", + " {\n", + " \"meanAveragePrecision\": 0.37167007,\n", + " \"confidenceMetrics\": [\n", + " {\n", + " \"precision\": 0.09565217,\n", + " \"f1Score\": 0.15985467,\n", + " \"confidenceThreshold\": 4.8275826e-05,\n", + " \"recall\": 0.48618785\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.0007978445,\n", + " \"f1Score\": 0.18373811,\n", + " \"precision\": 0.11357702,\n", + " \"recall\": 0.48066297\n", + " },\n", + " \n", + " # REMOVED FOR BREVITY\n", + " \n", + " {\n", + " \"f1Score\": 0.043243244,\n", + " \"precision\": 1.0,\n", + " \"confidenceThreshold\": 0.9953122,\n", + " \"recall\": 0.022099448\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99533135,\n", + " \"precision\": 1.0,\n", + " \"recall\": 0.016574586,\n", + " \"f1Score\": 0.032608695\n", + " },\n", + " {\n", + " \"f1Score\": 0.021857925,\n", + " \"confidenceThreshold\": 0.99550796,\n", + " \"precision\": 1.0,\n", + " \"recall\": 0.011049724\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.99619305,\n", + " \"f1Score\": 0.010989011,\n", + " \"precision\": 1.0,\n", + " \"recall\": 0.005524862\n", + " }\n", + " ],\n", + " \"iouThreshold\": 0.45,\n", + " \"meanAveragePrecision\": 0.32951096\n", + " }\n", + " ],\n", + " \"evaluatedBoundingBoxCount\": 181.0,\n", + " \"boundingBoxMeanAveragePrecision\": 0.2927905\n", + " },\n", + " \"createTime\": \"2021-02-26T03:38:42.086497Z\",\n", + " \"sliceDimensions\": [\n", + " \"annotationSpec\"\n", + " ]\n", + " }\n", + "]\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "models_evaluations_get:migration,new" + }, + "source": [ + "### [projects.locations.models.evaluations.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.models.evaluations/get)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "L3Sx6B8-V2HG" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "models_evaluations_get:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"model\"].get_model_evaluation(\n", + " name=evaluation_slice,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "fWpIdjpQV2HH" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "LQXJcdFQV2HH" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "oW7N6v-fS2ve" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/770273865954754560/evaluations/7557961565471768576\",\n", + " \"metricsSchemaUri\": \"gs://google-cloud-aiplatform/schema/modelevaluation/image_object_detection_metrics_1.0.0.yaml\",\n", + " \"metrics\": {\n", + " \"boundingBoxMeanAveragePrecision\": 0.2927905,\n", + " \"evaluatedBoundingBoxCount\": 181.0,\n", + " \"boundingBoxMetrics\": [\n", + " {\n", + " \"meanAveragePrecision\": 0.37167007,\n", + " \"confidenceMetrics\": [\n", + " {\n", + " \"f1Score\": 0.15985467,\n", + " \"confidenceThreshold\": 4.8275826e-05,\n", + " \"precision\": 0.09565217,\n", + " \"recall\": 0.48618785\n", + " },\n", + " {\n", + " \"precision\": 0.11357702,\n", + " \"confidenceThreshold\": 0.0007978445,\n", + " \"f1Score\": 0.18373811,\n", + " \"recall\": 0.48066297\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.003912397,\n", + " \"precision\": 0.15167548,\n", + " \"recall\": 0.47513813,\n", + " \"f1Score\": 0.22994651\n", + " },\n", + " \n", + " # REMOVED FOR BREVITY\n", + " \n", + " {\n", + " \"precision\": 1.0,\n", + " \"confidenceThreshold\": 0.9953122,\n", + " \"f1Score\": 0.043243244,\n", + " \"recall\": 0.022099448\n", + " },\n", + " {\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.032608695,\n", + " \"recall\": 0.016574586,\n", + " \"confidenceThreshold\": 0.99533135\n", + " },\n", + " {\n", + " \"recall\": 0.011049724,\n", + " \"precision\": 1.0,\n", + " \"confidenceThreshold\": 0.99550796,\n", + " \"f1Score\": 0.021857925\n", + " },\n", + " {\n", + " \"f1Score\": 0.010989011,\n", + " \"precision\": 1.0,\n", + " \"recall\": 0.005524862,\n", + " \"confidenceThreshold\": 0.99619305\n", + " }\n", + " ],\n", + " \"meanAveragePrecision\": 0.32951096,\n", + " \"iouThreshold\": 0.45\n", + " }\n", + " ]\n", + " },\n", + " \"createTime\": \"2021-02-26T03:38:42.086497Z\",\n", + " \"sliceDimensions\": [\n", + " \"annotationSpec\"\n", + " ]\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_predictions:migration" + }, + "source": [ + "## Make batch predictions\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:automl,image" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "Let's now make a batch input file, which you store in your local Cloud Storage bucket. The batch input file can be either CSV or JSONL. You will use JSONL in this tutorial. For JSONL file, you make one dictionary entry per line for each image. The dictionary contains the key/value pairs:\n", + "\n", + "- `content`: The Cloud Storage path to the image.\n", + "- `mimeType`: The content type. In our example, it is an `image/jpeg` file.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_items:automl,icn,csv" + }, + "outputs": [], + "source": [ + "test_items = ! gsutil cat $IMPORT_FILE | head -n2\n", + "\n", + "test_item_1, test_label_1 = test_items[0].split(\",\")[1], test_items[0].split(\",\")[2]\n", + "test_item_2, test_label_2 = test_items[0].split(\",\")[1], test_items[0].split(\",\")[2]\n", + "\n", + "file_1 = test_item_1.split(\"/\")[-1]\n", + "file_2 = test_item_2.split(\"/\")[-1]\n", + "\n", + "! gsutil cp $test_item_1 gs://$BUCKET_NAME/$file_1\n", + "! gsutil cp $test_item_2 gs://$BUCKET_NAME/$file_2\n", + "\n", + "test_item_1 = \"gs://\" + BUCKET_NAME + \"/\" + file_1\n", + "test_item_2 = \"gs://\" + BUCKET_NAME + \"/\" + file_2\n", + "\n", + "print(test_item_1, test_label_1)\n", + "print(test_item_2, test_label_2)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Vvt6o3FhS2v2" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210226015226/279324025_3e74a32a84_o.jpg Baked Goods\n", + "gs://migration-ucaip-trainingaip-20210226015226/279324025_3e74a32a84_o.jpg Baked Goods\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bxjbyhI3_jAW" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "import tensorflow as tf\n", + "\n", + "gcs_input_uri = \"gs://\" + BUCKET_NAME + \"/test.jsonl\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " data = {\"content\": test_item_1, \"mime_type\": \"image/jpeg\"}\n", + " f.write(json.dumps(data) + \"\\n\")\n", + " data = {\"content\": test_item_2, \"mime_type\": \"image/jpeg\"}\n", + " f.write(json.dumps(data) + \"\\n\")\n", + "\n", + "!gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "BsPpRX9ES2v3" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\"content\": \"gs://migration-ucaip-trainingaip-20210226015226/279324025_3e74a32a84_o.jpg\", \"mime_type\": \"image/jpeg\"}\n", + "{\"content\": \"gs://migration-ucaip-trainingaip-20210226015226/279324025_3e74a32a84_o.jpg\", \"mime_type\": \"image/jpeg\"}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_create:migration,new" + }, + "source": [ + "### [projects.locations.batchPredictionJobs.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.batchPredictionJobs/create)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "GT0I6yD7V2HN" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_create:migration,new,request,icn" + }, + "outputs": [], + "source": [ + "parameters = {\"confidenceThreshold\": 0.5, \"maxPredictions\": 2}\n", + "\n", + "batch_prediction_job = {\n", + " \"display_name\": \"salads_\" + TIMESTAMP,\n", + " \"model\": model_id,\n", + " \"input_config\": {\n", + " \"instances_format\": \"jsonl\",\n", + " \"gcs_source\": {\n", + " \"uris\": [gcs_input_uri],\n", + " },\n", + " },\n", + " \"model_parameters\": json_format.ParseDict(parameters, Value()),\n", + " \"output_config\": {\n", + " \"predictions_format\": \"jsonl\",\n", + " \"gcs_destination\": {\n", + " \"output_uri_prefix\": \"gs://\" + f\"{BUCKET_NAME}/batch_output/\",\n", + " },\n", + " },\n", + " \"dedicated_resources\": {\n", + " \"machine_spec\": {\n", + " \"machine_type\": \"n1-standard-2\",\n", + " \"accelerator_type\": 0,\n", + " },\n", + " \"starting_replica_count\": 1,\n", + " \"max_replica_count\": 1,\n", + " },\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateBatchPredictionJobRequest(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "1CiJ5MxmS2v4" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"batchPredictionJob\": {\n", + " \"displayName\": \"salads_20210226015226\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/770273865954754560\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210226015226/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"modelParameters\": {\n", + " \"confidenceThreshold\": 0.5,\n", + " \"maxPredictions\": 2.0\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210226015226/batch_output/\"\n", + " }\n", + " },\n", + " \"dedicatedResources\": {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-2\"\n", + " },\n", + " \"startingReplicaCount\": 1,\n", + " \"maxReplicaCount\": 1\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "_RXG0aaSV2HS" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].create_batch_prediction_job(\n", + " parent=PARENT,\n", + " batch_prediction_job=batch_prediction_job,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "LmvCsxd-V2HT" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "-w3S0zCGV2HU" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "rpBOX_k_S2v6" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/batchPredictionJobs/2404341658876379136\",\n", + " \"displayName\": \"salads_20210226015226\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/770273865954754560\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210226015226/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"modelParameters\": {\n", + " \"maxPredictions\": 2.0,\n", + " \"confidenceThreshold\": 0.5\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210226015226/batch_output/\"\n", + " }\n", + " },\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"completionStats\": {\n", + " \"incompleteCount\": \"-1\"\n", + " },\n", + " \"createTime\": \"2021-02-26T09:36:17.046416Z\",\n", + " \"updateTime\": \"2021-02-26T09:36:17.046416Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batch_job_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The fully qualified ID for the batch job\n", + "batch_job_id = request.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split(\"/\")[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_get:migration,new" + }, + "source": [ + "### [projects.locations.batchPredictionJobs.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.batchPredictionJobs/get)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "EDuJAyzbV2HW" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_get:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].get_batch_prediction_job(\n", + " name=batch_job_id,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "CHzK3m-8V2He" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "EJuTTf-NV2Hm" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "SdVXdHsES2v9" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/batchPredictionJobs/2404341658876379136\",\n", + " \"displayName\": \"salads_20210226015226\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/770273865954754560\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210226015226/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"modelParameters\": {\n", + " \"maxPredictions\": 2.0,\n", + " \"confidenceThreshold\": 0.5\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210226015226/batch_output/\"\n", + " }\n", + " },\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"completionStats\": {\n", + " \"incompleteCount\": \"-1\"\n", + " },\n", + " \"createTime\": \"2021-02-26T09:36:17.046416Z\",\n", + " \"updateTime\": \"2021-02-26T09:36:17.046416Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "def get_latest_predictions(gcs_out_dir):\n", + " \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n", + " folders = !gsutil ls $gcs_out_dir\n", + " latest = \"\"\n", + " for folder in folders:\n", + " subfolder = folder.split(\"/\")[-2]\n", + " if subfolder.startswith(\"prediction-\"):\n", + " if subfolder > latest:\n", + " latest = folder[:-1]\n", + " return latest\n", + "\n", + "\n", + "while True:\n", + " response = clients[\"job\"].get_batch_prediction_job(name=batch_job_id)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", response.state)\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " folder = get_latest_predictions(\n", + " response.output_config.gcs_destination.output_uri_prefix\n", + " )\n", + " ! gsutil ls $folder/prediction*.jsonl\n", + "\n", + " ! gsutil cat $folder/prediction*.jsonl\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "nGdOXkLKS2v-" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210226015226/batch_output/prediction-salads_20210226015226-2021-02-26T09:36:16.878261Z/predictions_00001.jsonl\n", + "{\"instance\":{\"content\":\"gs://migration-ucaip-trainingaip-20210226015226/279324025_3e74a32a84_o.jpg\",\"mimeType\":\"image/jpeg\"},\"prediction\":{\"ids\":[\"7754337640727445504\",\"8330798393030868992\"],\"displayNames\":[\"Salad\",\"Baked Goods\"],\"confidences\":[0.99217236,0.93992615],\"bboxes\":[[0.382205,0.9760891,0.29858154,0.9979937],[0.0012550354,0.5893767,0.06807296,0.81340706]]}}\n", + "{\"instance\":{\"content\":\"gs://migration-ucaip-trainingaip-20210226015226/279324025_3e74a32a84_o.jpg\",\"mimeType\":\"image/jpeg\"},\"prediction\":{\"ids\":[\"7754337640727445504\",\"8330798393030868992\"],\"displayNames\":[\"Salad\",\"Baked Goods\"],\"confidences\":[0.99217236,0.93992615],\"bboxes\":[[0.382205,0.9760891,0.29858154,0.9979937],[0.0012550354,0.5893767,0.06807296,0.81340706]]}}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "MSux-zY8eZxQ" + }, + "source": [ + "## Make online predictions" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "b-eWjz7feugD" + }, + "source": [ + "### Prepare file for online prediction" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tjM6rXqnfRRR" + }, + "source": [ + "### [projects.locations.endpoints.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "KYGpDPZAfYgU" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "WSz24F9a9Az4" + }, + "outputs": [], + "source": [ + "endpoint = {\n", + " \"display_name\": \"salads_\" + TIMESTAMP,\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateEndpointRequest(\n", + " parent=PARENT,\n", + " endpoint=endpoint,\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "-kZWsojqS2wA" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"endpoint\": {\n", + " \"displayName\": \"salads_20210226015226\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "_zboDSFUgLA0" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "leTjZnfQ9Az5" + }, + "outputs": [], + "source": [ + "request = clients[\"endpoint\"].create_endpoint(\n", + " parent=PARENT,\n", + " endpoint=endpoint,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "fU4aCBgngf7y" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gvVp3j7T9Az5" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "395MGwTGlJjw" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/endpoints/2449782275429105664\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "8Ufl6703g4aD" + }, + "source": [ + "### [projects.locations.endpoints.deployModel](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/deployModel)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ZR1fhICH9Az6" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "L8tvBPjZ9Az6" + }, + "outputs": [], + "source": [ + "deployed_model = {\n", + " \"model\": model_id,\n", + " \"display_name\": \"salads_\" + TIMESTAMP,\n", + " \"automatic_resources\": {\n", + " \"min_replica_count\": 1,\n", + " \"max_replica_count\": 1,\n", + " },\n", + "}\n", + "\n", + "traffic_split = {\n", + " \"0\": 100,\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.DeployModelRequest(\n", + " endpoint=endpoint_id,\n", + " deployed_model=deployed_model,\n", + " traffic_split=traffic_split,\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "GjkTPh0_S2wD" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"endpoint\": \"projects/116273516712/locations/us-central1/endpoints/2449782275429105664\",\n", + " \"deployedModel\": {\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/770273865954754560\",\n", + " \"displayName\": \"salads_20210226015226\",\n", + " \"automaticResources\": {\n", + " \"minReplicaCount\": 1,\n", + " \"maxReplicaCount\": 1\n", + " }\n", + " },\n", + " \"trafficSplit\": {\n", + " \"0\": 100\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "nZhbk8FH9Az6" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "80iu2TQa9Az6" + }, + "outputs": [], + "source": [ + "request = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint_id,\n", + " deployed_model=deployed_model,\n", + " traffic_split=traffic_split,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "xqYVmdRJ9Az7" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "kGSF1xLr9Az7" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "KEQXIKJWS2wE" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"deployedModel\": {\n", + " \"id\": \"3904304217581420544\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deployed_model_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The unique ID for the deployed model\n", + "deployed_model_id = result.deployed_model.id\n", + "\n", + "print(deployed_model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_predict:migration,new" + }, + "source": [ + "### [projects.locations.endpoints.predict](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/predict)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ozaMOwdNJjkT" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "5ZTgiJztS2wF" + }, + "outputs": [], + "source": [ + "import base64\n", + "\n", + "import tensorflow as tf\n", + "\n", + "single_file = ! gsutil cat $IMPORT_FILE | head -n 1\n", + "single_file = single_file[0].split(\",\")[1]\n", + "\n", + "with tf.io.gfile.GFile(single_file, \"rb\") as f:\n", + " content = f.read()\n", + "\n", + "instances_list = [{\"content\": base64.b64encode(content).decode(\"utf-8\")}]\n", + "instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + "parameters_dict = {\n", + " \"confidenceThreshold\": 0.5,\n", + " \"maxPredictions\": 2,\n", + "}\n", + "parameters = json_format.ParseDict(parameters_dict, Value())\n", + "\n", + "request = aip.PredictRequest(\n", + " endpoint=endpoint_id,\n", + " parameters=parameters,\n", + ")\n", + "request.instances.append(instances)\n", + "\n", + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_predict:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"endpoint\": \"projects/116273516712/locations/us-central1/endpoints/2449782275429105664\",\n", + " \"instances\": [\n", + " [\n", + " {\n", + " \"content\": \"/9j/4RtSRXhpZgAASUkqAAgAAAAIAA8BAgAGAAAAbgAAABABAgAEAAAATjkxABIBAwABAAAAAQAAABoBBQABAAAAdAAAABsBBQABAAAAfAAAACgBAwABAAAAAgAAABMCAwABAAAAAQAAAGmHBAABAAAAVh+71qaC5bdjJzRRdnuZVI80S+khxk0/zzXWpHmyjqPW4IHWnC755rWM3HW5DiH2jJzmke6296n22o+W4xtQIPWmrqJJPNVCr/X9MXJ1JPtW4Uvm+9N7iI3n9TTBcc9axqTa0EyObUUiXrVKTV0ZvvVjb\n", + "# REMOVED for brevity\n", + "KUtSNYfMbpVhYdpFZxXcL66F23TjcRnFTTzgLwabmuYbXUqlgDmoHmy2BVpqw29UVLqbqKxrqcKDVWtqFrGTe6l8pVelYF5KZmq7JvQ00iiFU5qxGTkfWtouzsS2f/Z\"\n", + " }\n", + " ]\n", + " ],\n", + " \"parameters\": {\n", + " \"confidenceThreshold\": 0.5,\n", + " \"maxPredictions\": 2.0\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "WSLQGw91S2wM" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_predict:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"prediction\"].predict(\n", + " endpoint=endpoint_id,\n", + " instances=instances,\n", + " parameters=parameters,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "NrLXrb0vS2wN" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Bb9oC1QqS2wO" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Jz9lq4SQS2wO" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"predictions\": [\n", + " {\n", + " \"ids\": [\n", + " \"7754337640727445504\",\n", + " \"8330798393030868992\"\n", + " ],\n", + " \"confidences\": [\n", + " 0.99217236,\n", + " 0.939926147\n", + " ],\n", + " \"displayNames\": [\n", + " \"Salad\",\n", + " \"Baked Goods\"\n", + " ],\n", + " \"bboxes\": [\n", + " [\n", + " 0.38220492,\n", + " 0.976089239,\n", + " 0.298581541,\n", + " 0.997993708\n", + " ],\n", + " [\n", + " 0.00125509501,\n", + " 0.589376688,\n", + " 0.0680729151,\n", + " 0.813407123\n", + " ]\n", + " ]\n", + " }\n", + " ],\n", + " \"deployedModelId\": \"3904304217581420544\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new" + }, + "source": [ + "### [projects.locations.endpoints.undeployModel](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/undeployModel)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "VrWkAw5ES2wP" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_undeploymodel:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint_id,\n", + " deployed_model_id=deployed_model_id,\n", + " traffic_split={},\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "io6vwLpAS2wQ" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "OuVvcaCdS2wQ" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup:migration,new" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "_eQuWWpLV2Hr" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_pipeline = True\n", + "delete_batchjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex AI fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset:\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex AI fully qualified identifier for the model\n", + "try:\n", + " if delete_model:\n", + " clients[\"model\"].delete_model(name=model_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex AI fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint:\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex AI fully qualified identifier for the training pipeline\n", + "try:\n", + " if delete_pipeline:\n", + " clients[\"pipeline\"].delete_training_pipeline(name=training_pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex AI fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob:\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r gs://$BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "collapsed_sections": [ + "gcp_authenticate", + "bucket:batch_prediction", + "import_aip", + "call:migration", + "p6gKpzGlV2GH", + "ZY3kU6P6V2GS", + "9wQgjU9AV2Gl", + "BfPthv3CV2G4", + "models_evaluations_list:migration,new", + "X47P3WEMV2HC", + "G7i7OmGRV2HD", + "models_evaluations_get:migration,new", + "L3Sx6B8-V2HG", + "fWpIdjpQV2HH", + "_RXG0aaSV2HS", + "EDuJAyzbV2HW" + ], + "name": "UJ5 unified AutoML Vision Video Classification.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/migration/UJ6 legacy AutoML Natural Language Text Classification.ipynb b/notebooks/community/migration/UJ6 legacy AutoML Natural Language Text Classification.ipynb new file mode 100644 index 000000000..c5008ce38 --- /dev/null +++ b/notebooks/community/migration/UJ6 legacy AutoML Natural Language Text Classification.ipynb @@ -0,0 +1,1691 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "8edU8FN2lH6o" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_title:migration,automl,icn" + }, + "source": [ + "# AutoML natural language text classification model\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "JE-aKjayJjkF" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of AutoML SDK." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "7dxB6-B0JjkG" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-automl" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Iddo-8HmJjkH" + }, + "source": [ + "Install the Google *cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "xHvi6EFSJjkH" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-storage" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ktRTgB8DJjkI" + }, + "source": [ + "### Restart the Kernel\n", + "\n", + "Once you've installed the AutoML SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "x1g4mBdlJjkI" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "if not os.getenv(\"AUTORUN\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Af0jTPSgl9yh" + }, + "source": [ + "## Before you begin\r\n", + "\r\n", + "### GPU run-time\r\n", + "\r\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\r\n", + "\r\n", + "### Set up your GCP project\r\n", + "\r\n", + "**The following steps are required, regardless of your notebook environment.**\r\n", + "\r\n", + "1. [Select or create a GCP project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\r\n", + "\r\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\r\n", + "\r\n", + "3. [Enable the AutoML APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\r\n", + "\r\n", + "4. [Google Cloud SDK](https://cloud.google.com/sdk) is already installed in AutoML Notebooks.\r\n", + "\r\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\r\n", + "Cloud SDK uses the right project for all the commands in this notebook.\r\n", + "\r\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "hceMthWGJjkI" + }, + "source": [ + "#### Project ID\n", + "\n", + "**If you don't know your project ID**, try to get your project ID using `gcloud` command by executing the second cell below." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Xzde8OBYDdh2" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "0UfsdLCEJjkJ" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "jT0fijejJjkK" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for AutoML. We recommend when possible, to choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You cannot use a Multi-Regional Storage bucket for training with AutoML. Not all regions provide support for all AutoML services. For the latest support per region, see [Region support for AutoML services]()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "CTBlncfrJjkK" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "J_croPBoJjkK" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "KKxStk5bJjkL" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "h_L3MRsOmYED" + }, + "source": [ + "### Authenticate your GCP account\r\n", + "\r\n", + "**If you are using AutoML Notebooks**, your environment is already\r\n", + "authenticated. Skip this step.\r\n", + "\r\n", + "*Note: If you are on an AutoML notebook and run the cell, the cell knows to skip executing the authentication steps.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "HvxYUdIfGAM-" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your Google Cloud account. This provides access\n", + "# to your Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Vertex, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this tutorial in a notebook locally, replace the string\n", + " # below with the path to your service account key and run this cell to\n", + " # authenticate your Google Cloud account.\n", + " else:\n", + " %env GOOGLE_APPLICATION_CREDENTIALS your_path_to_credentials.json\n", + "\n", + " # Log in to your account on Google Cloud\n", + " ! gcloud auth login" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "MDUAZaN3JjkL" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets. " + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n", + " BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "yU6VuylGJjkM" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "QVRixM4oJjkM" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Mi-Pdh6QJjkN" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aqDptKpYJjkN" + }, + "outputs": [], + "source": [ + "! gsutil ls -al gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "fR9geV9pJjkO" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "hIHTX-pkJjkO" + }, + "source": [ + "#### Import AutoML SDK\n", + "\n", + "Import the AutoM SDK into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "qrSkIKVgJjkO" + }, + "outputs": [], + "source": [ + "import json\n", + "import time\n", + "\n", + "from google.cloud import automl\n", + "from google.protobuf.json_format import MessageToJson" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "lbv411XjJjkP" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Setup up the following constants for AutoML:\n", + "\n", + "- `PARENT`: The AutoM location root path for dataset, model and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Gx8Eos8PJjkP" + }, + "outputs": [], + "source": [ + "# AutoM location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "JeM1smVJlH61" + }, + "source": [ + "## Clients\n", + "\n", + "The AutoML SDK works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the server (AutoML).\n", + "\n", + "You will use several clients in this tutorial, so set them all up upfront.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "c3sR3OxmxXiR" + }, + "outputs": [], + "source": [ + "def automl_client():\n", + " return automl.AutoMlClient()\n", + "\n", + "\n", + "def prediction_client():\n", + " return automl.PredictionServiceClient()\n", + "\n", + "\n", + "def operations_client():\n", + " return automl.AutoMlClient()._transport.operations_client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"automl\"] = automl_client()\n", + "clients[\"prediction\"] = prediction_client()\n", + "clients[\"operations\"] = operations_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:flowers,csv,icn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://cloud-ml-data/NL-classification/happiness.csv\"" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cjhb6UrAaGwi" + }, + "outputs": [], + "source": [ + "! gsutil cat $IMPORT_FILE | head -n 10" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gpsfIXYeaGwj" + }, + "source": [ + "*Example output*:\n", + "```\n", + "I went on a successful date with someone I felt sympathy and connection with.,affection\n", + "I was happy when my son got 90% marks in his examination,affection\n", + "I went to the gym this morning and did yoga.,exercise\n", + "We had a serious talk with some friends of ours who have been flaky lately. They understood and we had a good evening hanging out.,bonding\n", + "I went with grandchildren to butterfly display at Crohn Conservatory,affection\n", + "I meditated last night.,leisure\n", + "\"I made a new recipe for peasant bread, and it came out spectacular!\",achievement\n", + "I got gift from my elder brother which was really surprising me,affection\n", + "YESTERDAY MY MOMS BIRTHDAY SO I ENJOYED,enjoy_the_moment\n", + "Watching cupcake wars with my three teen children,affection\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_create_dataset:migration" + }, + "source": [ + "## Create a dataset" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_datasets_create:migration,old" + }, + "source": [ + "### [projects.locations.datasets.create](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.datasets/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ozaMOwdNJjkT" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "IFEbAUNwlH64" + }, + "outputs": [], + "source": [ + "dataset = {\n", + " \"display_name\": \"happiness_\" + TIMESTAMP,\n", + " \"text_classification_dataset_metadata\": {\"classification_type\": \"MULTICLASS\"},\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " automl.CreateDatasetRequest(parent=PARENT, dataset=dataset).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"dataset\": {\n", + " \"displayName\": \"happiness_20210228224317\",\n", + " \"textClassificationDatasetMetadata\": {\n", + " \"classificationType\": \"MULTICLASS\"\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "vHkuSZxkJjkT" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "o_JsNpQ1lH65" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].create_dataset(parent=PARENT, dataset=dataset)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "FLqkZMD2JjkT" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "5eOgT5MllH65" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "PGuI5t-gaGwm" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/datasets/TCN2705019056410329088\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wKsCzHBMaGwn" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_datasets_importdata:migration,old" + }, + "source": [ + "### [projects.locations.datasets.importData](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.datasets/importData)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "VkfdNI66lH66" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "MtS0VtlMlH66" + }, + "outputs": [], + "source": [ + "input_config = {\"gcs_source\": {\"input_uris\": [IMPORT_FILE]}}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " automl.ImportDataRequest(name=dataset_id, input_config=input_config).__dict__[\n", + " \"_pb\"\n", + " ]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Z52Bxts7aGwo" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/datasets/TCN2705019056410329088\",\n", + " \"inputConfig\": {\n", + " \"gcsSource\": {\n", + " \"inputUris\": [\n", + " \"gs://cloud-ml-data/NL-classification/happiness.csv\"\n", + " ]\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Gmg3fbX_lH66" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "yvEILSOelH66" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].import_data(name=dataset_id, input_config=input_config)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "WMUN_7EclH67" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Tf4Yf6JAlH67" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "6Pwgc1IYaGwq" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_create_and_deploy_model:migration" + }, + "source": [ + "## Train a model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_models_create:migration,old" + }, + "source": [ + "### [projects.locations.models.create](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sbJ0wIRHlH67" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "mCAqrD7elH68" + }, + "outputs": [], + "source": [ + "model = automl.Model(\n", + " display_name=\"happiness_\" + TIMESTAMP,\n", + " dataset_id=dataset_short_id,\n", + " text_classification_model_metadata=automl.TextClassificationModelMetadata(),\n", + ")\n", + "\n", + "print(\n", + " MessageToJson(automl.CreateModelRequest(parent=PARENT, model=model).__dict__[\"_pb\"])\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "PLAe7LgtaGwr" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"model\": {\n", + " \"displayName\": \"happiness_20210228224317\",\n", + " \"datasetId\": \"TCN2705019056410329088\",\n", + " \"textClassificationModelMetadata\": {}\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "K2UcVNlAlH68" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "FpSHTRfElH69" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].create_model(parent=PARENT, model=model)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "rTFtnIKZlH6-" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "UmoMM7VSlH6-" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "EzKxl38qaGws" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/TCN5333697920992542720\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "4QxYMtjYaGws" + }, + "outputs": [], + "source": [ + "# The full unique ID for the training pipeline\n", + "model_id = result.name\n", + "# The short numeric ID for the training pipeline\n", + "model_short_id = model_id.split(\"/\")[-1]\n", + "\n", + "print(model_short_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_evaluate_the_model:migration" + }, + "source": [ + "## Evaluate the model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "yfufMwAEJjkX" + }, + "source": [ + "### [projects.locations.models.modelEvaluations.list](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models.modelEvaluations/list)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "4wJgA2D8lH6_" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "fxMsWaailH6_" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].list_model_evaluations(parent=model_id, filter=\"\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "IBOLi2gflH6_" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "E5xVerw2aGwu" + }, + "outputs": [], + "source": [ + "evaluations_list = [\n", + " json.loads(MessageToJson(me.__dict__[\"_pb\"])) for me in request.model_evaluation\n", + "]\n", + "\n", + "print(json.dumps(evaluations_list, indent=2))\n", + "# The evaluation slice\n", + "evaluation_slice = request.model_evaluation[0].name" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bB3px5SBxXig" + }, + "source": [ + "*Example output*:\n", + "```\n", + "[\n", + " {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/TCN5333697920992542720/modelEvaluations/1436745357261371663\",\n", + " \"annotationSpecId\": \"3130761503557287936\",\n", + " \"createTime\": \"2021-03-01T02:56:28.878044Z\",\n", + " \"evaluatedExampleCount\": 1193,\n", + " \"classificationEvaluationMetrics\": {\n", + " \"auPrc\": 0.99065405,\n", + " \"confidenceMetricsEntry\": [\n", + " {\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.01424979,\n", + " \"f1Score\": 0.028099174\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.05,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.5862069,\n", + " \"f1Score\": 0.73913044\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.94,\n", + " \"recall\": 0.64705884,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.7857143\n", + " },\n", + " \n", + " # REMOVED FOR BREVITY\n", + " \n", + " {\n", + " \"confidenceThreshold\": 0.999,\n", + " \"recall\": 0.21372032,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.35217392\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 1.0,\n", + " \"recall\": 0.0026385225,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.005263158\n", + " }\n", + " ],\n", + " \"logLoss\": 0.14686257\n", + " },\n", + " \"displayName\": \"achievement\"\n", + " }\n", + "]\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "i6T0bzuNJjkY" + }, + "source": [ + "### [projects.locations.models.modelEvaluations.get](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models.modelEvaluations/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "16ZPQWjulH6_" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aorkRa76lH7A" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].get_model_evaluation(name=evaluation_slice)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pHodXIWxlH7A" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "CX5GXZkDlH7A" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "XauXDBQfaGwx" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/TCN5333697920992542720/modelEvaluations/1436745357261371663\",\n", + " \"annotationSpecId\": \"3130761503557287936\",\n", + " \"createTime\": \"2021-03-01T02:56:28.878044Z\",\n", + " \"evaluatedExampleCount\": 1193,\n", + " \"classificationEvaluationMetrics\": {\n", + " \"auPrc\": 0.99065405,\n", + " \"confidenceMetricsEntry\": [\n", + " {\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.01424979,\n", + " \"f1Score\": 0.028099174\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.05,\n", + " \"recall\": 1.0,\n", + " \"precision\": 0.5862069,\n", + " \"f1Score\": 0.73913044\n", + " },\n", + " \n", + " # REMOVED FOR BREVITY\n", + " \n", + " {\n", + " \"confidenceThreshold\": 0.999,\n", + " \"recall\": 0.23529412,\n", + " \"precision\": 1.0,\n", + " \"f1Score\": 0.3809524\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 1.0,\n", + " \"precision\": 1.0\n", + " }\n", + " ],\n", + " \"logLoss\": 0.005436425\n", + " },\n", + " \"displayName\": \"exercise\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "xksNVRJIlH7A" + }, + "source": [ + "## Make batch predictions" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "CilSsXsyaGw5" + }, + "source": [ + "### Prepare files for batch prediction" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "6uSIK0_jaGw5" + }, + "outputs": [], + "source": [ + "test_item = ! gsutil cat $IMPORT_FILE | head -n1\n", + "test_item, test_label = str(test_item[0]).split(\",\")\n", + "\n", + "print(test_item, test_label)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "4Saj5dsWlH7A" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "import tensorflow as tf\n", + "\n", + "test_item_uri = \"gs://\" + BUCKET_NAME + \"/test.txt\"\n", + "with tf.io.gfile.GFile(test_item_uri, \"w\") as f:\n", + " f.write(test_item + \"\\n\")\n", + "\n", + "gcs_input_uri = \"gs://\" + BUCKET_NAME + \"/batch.csv\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " f.write(test_item_uri + \"\\n\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "nTTzKkr7aGw6" + }, + "outputs": [], + "source": [ + "! gsutil cat $gcs_input_uri\n", + "! gsutil cat $test_item_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "TPtwEmoeaGw6" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210228224317/test.txt\n", + "I went on a successful date with someone I felt sympathy and connection with.\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_models_batchpredict:migration,old" + }, + "source": [ + "### [projects.locations.models.batchPredict](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models/batchPredict)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "9Vyk6FVxlH7C" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "C73-3tt0lH7C" + }, + "outputs": [], + "source": [ + "input_config = {\"gcs_source\": {\"input_uris\": [gcs_input_uri]}}\n", + "\n", + "output_config = {\n", + " \"gcs_destination\": {\"output_uri_prefix\": \"gs://\" + f\"{BUCKET_NAME}/batch_output/\"}\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " automl.BatchPredictRequest(\n", + " name=model_id, input_config=input_config, output_config=output_config\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "TX0IQkdOaGw8" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/TCN5333697920992542720\",\n", + " \"inputConfig\": {\n", + " \"gcsSource\": {\n", + " \"inputUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210228224317/batch.csv\"\n", + " ]\n", + " }\n", + " },\n", + " \"outputConfig\": {\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210228224317/batch_output/\"\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "D5pvxloAlH7C" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "VMiC645dlH7C" + }, + "outputs": [], + "source": [ + "request = clients[\"prediction\"].batch_predict(\n", + " name=model_id, input_config=input_config, output_config=output_config\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "kFtsjWHIlH7D" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Nqg__59clH7D" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "771dDuKzg8Mk" + }, + "source": [ + "*Example output*:\r\n", + "```\r\n", + "{}\r\n", + "```" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "fnYRCde3xXie" + }, + "outputs": [], + "source": [ + "destination_uri = output_config[\"gcs_destination\"][\"output_uri_prefix\"][:-1]\n", + "\n", + "! gsutil ls $destination_uri/*\n", + "! gsutil cat $destination_uri/prediction*/*.jsonl" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gVNVzw5EaGw-" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210228224317/batch_output/prediction-happiness_20210228224317-2021-03-01T02:57:02.004934Z/text_classification_1.jsonl\n", + "gs://migration-ucaip-trainingaip-20210228224317/batch_output/prediction-happiness_20210228224317-2021-03-01T02:57:02.004934Z/text_classification_2.jsonl\n", + "{\"textSnippet\":{\"contentUri\":\"gs://migration-ucaip-trainingaip-20210228224317/test.txt\"},\"annotations\":[{\"annotationSpecId\":\"5436604512770981888\",\"classification\":{\"score\":0.93047273},\"displayName\":\"affection\"},{\"annotationSpecId\":\"3707222255860711424\",\"classification\":{\"score\":0.002518793},\"displayName\":\"achievement\"},{\"annotationSpecId\":\"7742447521984675840\",\"classification\":{\"score\":1.3182563E-4},\"displayName\":\"enjoy_the_moment\"},{\"annotationSpecId\":\"824918494343593984\",\"classification\":{\"score\":0.06613126},\"displayName\":\"bonding\"},{\"annotationSpecId\":\"1977839998950440960\",\"classification\":{\"score\":1.5267624E-5},\"displayName\":\"leisure\"},{\"annotationSpecId\":\"8318908274288099328\",\"classification\":{\"score\":8.887557E-6},\"displayName\":\"nature\"},{\"annotationSpecId\":\"3130761503557287936\",\"classification\":{\"score\":7.2130124E-4},\"displayName\":\"exercise\"}]}\n", + "```" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "y_rNqE85lH7D" + }, + "source": [ + "## Make online predictions" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_models_deploy:migration,old" + }, + "source": [ + "### [projects.locations.models.deploy](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models/deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "VUfhwbkLlH7E" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tojYy4j6lH7E" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].deploy_model(name=model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "8OaGY7JmlH7E" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "voC8sroOlH7E" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "CZ-62obNmBNc" + }, + "source": [ + "*Example output*:\r\n", + "```\r\n", + "{}\r\n", + "```" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_models_predict:migration,old" + }, + "source": [ + "### [projects.locations.models.predict](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models/predict)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "v3Hc1PEyyabD" + }, + "source": [ + "### Prepare data item for online prediction" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "EzXouIJ0aGxB" + }, + "outputs": [], + "source": [ + "test_item = ! gsutil cat $IMPORT_FILE | head -n1\n", + "test_item, test_label = str(test_item[0]).split(\",\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "LcSHb_A0lH7F" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "BEHtyu-BlH7F" + }, + "outputs": [], + "source": [ + "payload = {\"text_snippet\": {\"content\": test_item, \"mime_type\": \"text/plain\"}}\n", + "\n", + "request = automl.PredictRequest(name=model_id, payload=payload)\n", + "\n", + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "In8qh2Z1aGxC" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/TCN5333697920992542720\",\n", + " \"payload\": {\n", + " \"textSnippet\": {\n", + " \"content\": \"I went on a successful date with someone I felt sympathy and connection with.\",\n", + " \"mimeType\": \"text/plain\"\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "xR4xdzeflH7F" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Cfy8J23SlH7F" + }, + "outputs": [], + "source": [ + "request = clients[\"prediction\"].predict(request=request)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "N0dT5DXblH7F" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "CHf-Y_XtlH7G" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "3RxSK0xDaGxC" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"payload\": [\n", + " {\n", + " \"annotationSpecId\": \"5436604512770981888\",\n", + " \"classification\": {\n", + " \"score\": 0.9272586\n", + " },\n", + " \"displayName\": \"affection\"\n", + " },\n", + " {\n", + " \"annotationSpecId\": \"824918494343593984\",\n", + " \"classification\": {\n", + " \"score\": 0.068884976\n", + " },\n", + " \"displayName\": \"bonding\"\n", + " },\n", + " {\n", + " \"annotationSpecId\": \"3707222255860711424\",\n", + " \"classification\": {\n", + " \"score\": 0.0028119811\n", + " },\n", + " \"displayName\": \"achievement\"\n", + " },\n", + " {\n", + " \"annotationSpecId\": \"3130761503557287936\",\n", + " \"classification\": {\n", + " \"score\": 0.0008869726\n", + " },\n", + " \"displayName\": \"exercise\"\n", + " },\n", + " {\n", + " \"annotationSpecId\": \"7742447521984675840\",\n", + " \"classification\": {\n", + " \"score\": 0.00013229548\n", + " },\n", + " \"displayName\": \"enjoy_the_moment\"\n", + " },\n", + " {\n", + " \"annotationSpecId\": \"1977839998950440960\",\n", + " \"classification\": {\n", + " \"score\": 1.5584701e-05\n", + " },\n", + " \"displayName\": \"leisure\"\n", + " },\n", + " {\n", + " \"annotationSpecId\": \"8318908274288099328\",\n", + " \"classification\": {\n", + " \"score\": 9.5975e-06\n", + " },\n", + " \"displayName\": \"nature\"\n", + " }\n", + " ]\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bQ-VVaSxJjkd" + }, + "source": [ + "# Cleaning up\r\n", + "\r\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\r\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\r\n", + "\r\n", + "Otherwise, you can delete the individual resources you created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pTymPfKolH7G" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_model = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the AutoML fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset:\n", + " clients[\"automl\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the AutoML fully qualified identifier for the model\n", + "try:\n", + " if delete_model:\n", + " clients[\"automl\"].delete_model(name=model_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r gs://$BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "collapsed_sections": [ + "J_croPBoJjkK", + "h_L3MRsOmYED", + "MDUAZaN3JjkL", + "fR9geV9pJjkO", + "hIHTX-pkJjkO", + "lbv411XjJjkP", + "vHkuSZxkJjkT", + "FLqkZMD2JjkT", + "text_datasets_importdata:migration,old", + "VkfdNI66lH66", + "Gmg3fbX_lH66", + "WMUN_7EclH67", + "text_models_create:migration,old", + "sbJ0wIRHlH67", + "K2UcVNlAlH68", + "rTFtnIKZlH6-", + "yfufMwAEJjkX", + "4wJgA2D8lH6_", + "IBOLi2gflH6_", + "i6T0bzuNJjkY", + "16ZPQWjulH6_", + "pHodXIWxlH7A", + "CilSsXsyaGw5", + "text_models_batchpredict:migration,old", + "9Vyk6FVxlH7C", + "D5pvxloAlH7C", + "kFtsjWHIlH7D", + "text_models_deploy:migration,old", + "VUfhwbkLlH7E", + "8OaGY7JmlH7E", + "text_models_predict:migration,old", + "v3Hc1PEyyabD", + "LcSHb_A0lH7F", + "xR4xdzeflH7F", + "N0dT5DXblH7F" + ], + "name": "UJ6 legacy AutoML Natural Language Text Classification.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/migration/UJ6 unified AutoML Natural Language Text Classification.ipynb b/notebooks/community/migration/UJ6 unified AutoML Natural Language Text Classification.ipynb new file mode 100644 index 000000000..96e65293e --- /dev/null +++ b/notebooks/community/migration/UJ6 unified AutoML Natural Language Text Classification.ipynb @@ -0,0 +1,2535 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bLRZTKFkDdht" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "AKoTR-m2PmAR" + }, + "source": [ + "# Vertex SDK: AutoML natural language text classification model\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "JE-aKjayJjkF" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest (preview) version of Vertex SDK." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "7dxB6-B0JjkG" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-aiplatform --user" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Iddo-8HmJjkH" + }, + "source": [ + "Install the Google *cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "xHvi6EFSJjkH" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-storage" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ktRTgB8DJjkI" + }, + "source": [ + "### Restart the Kernel\n", + "\n", + "Once you've installed the Vertex SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "x1g4mBdlJjkI" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "if not os.getenv(\"AUTORUN\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "hceMthWGJjkI" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU run-time\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your GCP project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a GCP project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebooks.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Xzde8OBYDdh2" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "0UfsdLCEJjkJ" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "jT0fijejJjkK" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex AI. We recommend when possible, to choose the region closest to you. \n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You cannot use a Multi-Regional Storage bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see [Region support for Vertex AI services](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "CTBlncfrJjkK" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "J_croPBoJjkK" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "KKxStk5bJjkL" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0yDWDaKwF48e" + }, + "source": [ + "### Authenticate your GCP account\r\n", + "\r\n", + "**If you are using Google Cloud Notebooks**, your environment is already\r\n", + "authenticated. Skip this step.\r\n", + "\r\n", + "*Note: If you are on an Vertex notebook and run the cell, the cell knows to skip executing the authentication steps.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "HvxYUdIfGAM-" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your Google Cloud account. This provides access\n", + "# to your Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Vertex, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this tutorial in a notebook locally, replace the string\n", + " # below with the path to your service account key and run this cell to\n", + " # authenticate your Google Cloud account.\n", + " else:\n", + " %env GOOGLE_APPLICATION_CREDENTIALS your_path_to_credentials.json\n", + "\n", + " # Log in to your account on Google Cloud\n", + " ! gcloud auth login" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "MDUAZaN3JjkL" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets. " + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n", + " BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "yU6VuylGJjkM" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "QVRixM4oJjkM" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Mi-Pdh6QJjkN" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aqDptKpYJjkN" + }, + "outputs": [], + "source": [ + "! gsutil ls -al gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "fR9geV9pJjkO" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "hIHTX-pkJjkO" + }, + "source": [ + "#### Import Vertex SDK\n", + "\n", + "Import the Vertex SDK into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "qrSkIKVgJjkO" + }, + "outputs": [], + "source": [ + "import base64\n", + "import json\n", + "import os\n", + "import sys\n", + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "lbv411XjJjkP" + }, + "source": [ + "#### Vertex AI constants\n", + "\n", + "Setup up the following constants for Vertex AI:\n", + "\n", + "- `API_ENDPOINT`: The Vertex AI API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex AI location root path for dataset, model and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Gx8Eos8PJjkP" + }, + "outputs": [], + "source": [ + "# API Endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex AI location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "4x_t-MWnJjkQ" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Next, setup constants unique to AutoML Text Classification datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the managed dataset service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the managed dataset service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the managed pipelines service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "nmzhH42tJjkQ" + }, + "outputs": [], + "source": [ + "# Text Dataset type\n", + "TEXT_SCHEMA = \"google-cloud-aiplatform/schema/dataset/metadata/text_1.0.0.yaml\"\n", + "# Text Labeling type\n", + "IMPORT_SCHEMA_TEXT_CLASSIFICATION = \"gs://google-cloud-aiplatform/schema/dataset/ioformat/text_classification_single_label_io_format_1.0.0.yaml\"\n", + "# Text Training task\n", + "TRAINING_TEXT_CLASSIFICATION_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_text_classification_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "R7P3MRNgJjkQ" + }, + "source": [ + "## Clients Vertex AI\n", + "\n", + "The Vertex SDK works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the server (Vertex).\n", + "\n", + "You will use several clients in this tutorial, so set them all up upfront.\n", + "\n", + "- Dataset Service for managed datasets.\n", + "- Model Service for managed models.\n", + "- Pipeline Service for training.\n", + "- Endpoint Service for deployment.\n", + "- Prediction Service for serving. *Note*: Prediction has a different service endpoint." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "kt8j_ey4JjkR" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "clients[\"job\"] = create_job_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Ii4kquRRPmAn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://cloud-ml-data/NL-classification/happiness.csv\"" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ATVQWpgnPmAo" + }, + "outputs": [], + "source": [ + "! gsutil cat $IMPORT_FILE | head -n 10" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "I went on a successful date with someone I felt sympathy and connection with.,affection\n", + "I was happy when my son got 90% marks in his examination,affection\n", + "I went to the gym this morning and did yoga.,exercise\n", + "We had a serious talk with some friends of ours who have been flaky lately. They understood and we had a good evening hanging out.,bonding\n", + "I went with grandchildren to butterfly display at Crohn Conservatory,affection\n", + "I meditated last night.,leisure\n", + "\"I made a new recipe for peasant bread, and it came out spectacular!\",achievement\n", + "I got gift from my elder brother which was really surprising me,affection\n", + "YESTERDAY MY MOMS BIRTHDAY SO I ENJOYED,enjoy_the_moment\n", + "Watching cupcake wars with my three teen children,affection\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_create_dataset:migration" + }, + "source": [ + "## Create a dataset" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_datasets_create:migration,new" + }, + "source": [ + "### [projects.locations.datasets.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.datasets/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "L9TCLZcqDdh-" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "LV0CGdWxDdh-" + }, + "outputs": [], + "source": [ + "DATA_SCHEMA = TEXT_SCHEMA\n", + "\n", + "dataset = {\n", + " \"display_name\": \"happiness_\" + TIMESTAMP,\n", + " \"metadata_schema_uri\": \"gs://\" + DATA_SCHEMA,\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateDatasetRequest(parent=PARENT, dataset=dataset).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "oW46N7zDPmAq" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"dataset\": {\n", + " \"displayName\": \"happiness_20210226015238\",\n", + " \"metadataSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/metadata/text_1.0.0.yaml\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "46JjhmPfDdh_" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Opc77NQcDdh_" + }, + "outputs": [], + "source": [ + "request = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "6rltMZbbDdh_" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "zIJgoVVpDdh_" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "1EbxR8zZPmAt" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/datasets/574578388396670976\",\n", + " \"displayName\": \"happiness_20210226015238\",\n", + " \"metadataSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/metadata/text_1.0.0.yaml\",\n", + " \"labels\": {\n", + " \"aiplatform.googleapis.com/dataset_metadata_schema\": \"TEXT\"\n", + " },\n", + " \"metadata\": {\n", + " \"dataItemSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/dataitem/text_1.0.0.yaml\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "GiCxvMfDPmAu" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_datasets_import:migration,new" + }, + "source": [ + "### [projects.locations.datasets.import](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.datasets/import)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "rTHSmwyHDdh_" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bt8QqcrQDdiA" + }, + "outputs": [], + "source": [ + "LABEL_SCHEMA = IMPORT_SCHEMA_TEXT_CLASSIFICATION\n", + "\n", + "import_config = {\n", + " \"gcs_source\": {\"uris\": [IMPORT_FILE]},\n", + " \"import_schema_uri\": LABEL_SCHEMA,\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.ImportDataRequest(\n", + " name=dataset_short_id, import_configs=[import_config]\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Jjzp5G2QPmAw" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"574578388396670976\",\n", + " \"importConfigs\": [\n", + " {\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://cloud-ml-data/NL-classification/happiness.csv\"\n", + " ]\n", + " },\n", + " \"importSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/ioformat/text_classification_single_label_io_format_1.0.0.yaml\"\n", + " }\n", + " ]\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "BuJAHvs4DdiA" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "XccT_J1CDdiA" + }, + "outputs": [], + "source": [ + "request = clients[\"dataset\"].import_data(\n", + " name=dataset_id, import_configs=[import_config]\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bh6rq24uDdiA" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "skZIZaulDdiA" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "drQ196C7PmA0" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_create_and_deploy_model:migration" + }, + "source": [ + "## Train a model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0oqIBOSnJjkW" + }, + "source": [ + "### [projects.locations.trainingPipelines.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.trainingPipelines/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "6TEExyzrDdiB" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "IqU_AskkDdiB" + }, + "outputs": [], + "source": [ + "TRAINING_SCHEMA = TRAINING_TEXT_CLASSIFICATION_SCHEMA\n", + "\n", + "task = json_format.ParseDict(\n", + " {\n", + " \"multi_label\": False,\n", + " },\n", + " Value(),\n", + ")\n", + "\n", + "training_pipeline = {\n", + " \"display_name\": \"happiness_\" + TIMESTAMP,\n", + " \"input_data_config\": {\"dataset_id\": dataset_short_id},\n", + " \"model_to_upload\": {\"display_name\": \"happiness_\" + TIMESTAMP},\n", + " \"training_task_definition\": TRAINING_SCHEMA,\n", + " \"training_task_inputs\": task,\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateTrainingPipelineRequest(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "FRJplFjTPmA2" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"trainingPipeline\": {\n", + " \"displayName\": \"happiness_20210226015238\",\n", + " \"inputDataConfig\": {\n", + " \"datasetId\": \"574578388396670976\"\n", + " },\n", + " \"trainingTaskDefinition\": \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_text_classification_1.0.0.yaml\",\n", + " \"trainingTaskInputs\": {\n", + " \"multi_label\": false\n", + " },\n", + " \"modelToUpload\": {\n", + " \"displayName\": \"happiness_20210226015238\"\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "VeO6x5u6DdiC" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "oehjkwElDdiC" + }, + "outputs": [], + "source": [ + "request = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "b6D1PnMUDdiC" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "qbwBVg9NDdiC" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "qJY27GO1PmA3" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/trainingPipelines/2903115317607661568\",\n", + " \"displayName\": \"happiness_20210226015238\",\n", + " \"inputDataConfig\": {\n", + " \"datasetId\": \"574578388396670976\"\n", + " },\n", + " \"trainingTaskDefinition\": \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_text_classification_1.0.0.yaml\",\n", + " \"trainingTaskInputs\": {},\n", + " \"modelToUpload\": {\n", + " \"displayName\": \"happiness_20210226015238\"\n", + " },\n", + " \"state\": \"PIPELINE_STATE_PENDING\",\n", + " \"createTime\": \"2021-02-26T02:23:54.166560Z\",\n", + " \"updateTime\": \"2021-02-26T02:23:54.166560Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "R3dVRAJ2PmA4" + }, + "outputs": [], + "source": [ + "# The full unique ID for the training pipeline\n", + "training_pipeline_id = request.name\n", + "# The short numeric ID for the training pipeline\n", + "training_pipeline_short_id = training_pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(training_pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "QJsV1xoSJjkW" + }, + "source": [ + "### [projects.locations.trainingPipelines.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.trainingPipelines/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "jMx7wPO4DdiC" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "nkSMNyEHDdiD" + }, + "outputs": [], + "source": [ + "request = clients[\"pipeline\"].get_training_pipeline(name=training_pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bnn0DVfVDdiD" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "RxwZkKHHDdiD" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "W3vyjjnIPmA6" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/trainingPipelines/2903115317607661568\",\n", + " \"displayName\": \"happiness_20210226015238\",\n", + " \"inputDataConfig\": {\n", + " \"datasetId\": \"574578388396670976\"\n", + " },\n", + " \"trainingTaskDefinition\": \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_text_classification_1.0.0.yaml\",\n", + " \"trainingTaskInputs\": {},\n", + " \"modelToUpload\": {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/2369051733671280640\",\n", + " \"displayName\": \"happiness_20210226015238\"\n", + " },\n", + " \"state\": \"PIPELINE_STATE_SUCCEEDED\",\n", + " \"createTime\": \"2021-02-26T02:23:54.166560Z\",\n", + " \"startTime\": \"2021-02-26T02:23:54.396088Z\",\n", + " \"endTime\": \"2021-02-26T06:08:06.548524Z\",\n", + " \"updateTime\": \"2021-02-26T06:08:06.548524Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=training_pipeline_id)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " model_to_deploy_name = None\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " break\n", + " else:\n", + " model_id = response.model_to_upload.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(20)\n", + "\n", + "print(model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_evaluate_the_model:migration" + }, + "source": [ + "## Evaluate the model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "BJNzhELXJjkY" + }, + "source": [ + "### [projects.locations.models.evaluations.list](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.models.evaluations/list)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "--qyaaviDdiG" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "q5c2I_oFDdiH" + }, + "outputs": [], + "source": [ + "request = clients[\"model\"].list_model_evaluations(parent=model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "njIDMGb0DdiH" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "y0wUwEgUDdiH" + }, + "outputs": [], + "source": [ + "model_evaluations = [json.loads(MessageToJson(mel.__dict__[\"_pb\"])) for mel in request]\n", + "\n", + "print(json.dumps(model_evaluations, indent=2))\n", + "\n", + "# The evaluation slice\n", + "evaluation_slice = request.model_evaluations[0].name" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0RsA8RqKPmA-" + }, + "source": [ + "*Example output*:\n", + "```\n", + "[\n", + " {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/2369051733671280640/evaluations/1541152463304785920\",\n", + " \"metricsSchemaUri\": \"gs://google-cloud-aiplatform/schema/modelevaluation/classification_metrics_1.0.0.yaml\",\n", + " \"metrics\": {\n", + " \"confusionMatrix\": {\n", + " \"annotationSpecs\": [\n", + " {\n", + " \"displayName\": \"exercise\",\n", + " \"id\": \"952213353537732608\"\n", + " },\n", + " {\n", + " \"id\": \"1528674105841156096\",\n", + " \"displayName\": \"achievement\"\n", + " },\n", + " {\n", + " \"id\": \"3258056362751426560\",\n", + " \"displayName\": \"leisure\"\n", + " },\n", + " {\n", + " \"id\": \"3834517115054850048\",\n", + " \"displayName\": \"bonding\"\n", + " },\n", + " {\n", + " \"id\": \"5563899371965120512\",\n", + " \"displayName\": \"enjoy_the_moment\"\n", + " },\n", + " {\n", + " \"id\": \"6140360124268544000\",\n", + " \"displayName\": \"nature\"\n", + " },\n", + " {\n", + " \"id\": \"8446203133482237952\",\n", + " \"displayName\": \"affection\"\n", + " }\n", + " ],\n", + " \"rows\": [\n", + " [\n", + " 19.0,\n", + " 1.0,\n", + " 0.0,\n", + " 0.0,\n", + " 0.0,\n", + " 0.0,\n", + " 0.0\n", + " ],\n", + " [\n", + " 0.0,\n", + " 342.0,\n", + " 5.0,\n", + " 2.0,\n", + " 13.0,\n", + " 2.0,\n", + " 13.0\n", + " ],\n", + " [\n", + " 2.0,\n", + " 10.0,\n", + " 42.0,\n", + " 1.0,\n", + " 12.0,\n", + " 0.0,\n", + " 2.0\n", + " ],\n", + " [\n", + " 0.0,\n", + " 4.0,\n", + " 0.0,\n", + " 121.0,\n", + " 1.0,\n", + " 0.0,\n", + " 4.0\n", + " ],\n", + " [\n", + " 2.0,\n", + " 29.0,\n", + " 3.0,\n", + " 2.0,\n", + " 98.0,\n", + " 0.0,\n", + " 6.0\n", + " ],\n", + " [\n", + " 0.0,\n", + " 3.0,\n", + " 0.0,\n", + " 1.0,\n", + " 0.0,\n", + " 21.0,\n", + " 1.0\n", + " ],\n", + " [\n", + " 0.0,\n", + " 7.0,\n", + " 0.0,\n", + " 1.0,\n", + " 6.0,\n", + " 0.0,\n", + " 409.0\n", + " ]\n", + " ]\n", + " },\n", + " \"confidenceMetrics\": [\n", + " {\n", + " \"f1Score\": 0.25,\n", + " \"recall\": 1.0,\n", + " \"f1ScoreAt1\": 0.88776374,\n", + " \"precisionAt1\": 0.88776374,\n", + " \"precision\": 0.14285715,\n", + " \"recallAt1\": 0.88776374\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.05,\n", + " \"recall\": 0.9721519,\n", + " \"f1Score\": 0.8101266,\n", + " \"recallAt1\": 0.88776374,\n", + " \"f1ScoreAt1\": 0.88776374,\n", + " \"precisionAt1\": 0.88776374,\n", + " \"precision\": 0.69439423\n", + " },\n", + " \n", + " # REMOVED FOR BREVITY\n", + " \n", + " {\n", + " \"f1Score\": 0.0033698399,\n", + " \"recall\": 0.0016877637,\n", + " \"confidenceThreshold\": 1.0,\n", + " \"recallAt1\": 0.0016877637,\n", + " \"f1ScoreAt1\": 0.0033698399,\n", + " \"precisionAt1\": 1.0,\n", + " \"precision\": 1.0\n", + " }\n", + " ],\n", + " \"auPrc\": 0.95903283,\n", + " \"logLoss\": 0.08260541\n", + " },\n", + " \"createTime\": \"2021-02-26T06:07:48.967028Z\",\n", + " \"sliceDimensions\": [\n", + " \"annotationSpec\"\n", + " ]\n", + " }\n", + "]\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wr2JuU5nJjka" + }, + "source": [ + "### [projects.locations.models.evaluations.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.models.evaluations/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "edJ4XjSbDdiI" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "8R-P7LzdDdiI" + }, + "outputs": [], + "source": [ + "request = clients[\"model\"].get_model_evaluation(name=evaluation_slice)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "nyTwMNrmDdiI" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wj_LtYzJDdiI" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cjAR1CVDPmBB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/2369051733671280640/evaluations/1541152463304785920\",\n", + " \"metricsSchemaUri\": \"gs://google-cloud-aiplatform/schema/modelevaluation/classification_metrics_1.0.0.yaml\",\n", + " \"metrics\": {\n", + " \"confusionMatrix\": {\n", + " \"annotationSpecs\": [\n", + " {\n", + " \"displayName\": \"exercise\",\n", + " \"id\": \"952213353537732608\"\n", + " },\n", + " {\n", + " \"displayName\": \"achievement\",\n", + " \"id\": \"1528674105841156096\"\n", + " },\n", + " {\n", + " \"id\": \"3258056362751426560\",\n", + " \"displayName\": \"leisure\"\n", + " },\n", + " {\n", + " \"id\": \"3834517115054850048\",\n", + " \"displayName\": \"bonding\"\n", + " },\n", + " {\n", + " \"displayName\": \"enjoy_the_moment\",\n", + " \"id\": \"5563899371965120512\"\n", + " },\n", + " {\n", + " \"displayName\": \"nature\",\n", + " \"id\": \"6140360124268544000\"\n", + " },\n", + " {\n", + " \"id\": \"8446203133482237952\",\n", + " \"displayName\": \"affection\"\n", + " }\n", + " ],\n", + " \"rows\": [\n", + " [\n", + " 19.0,\n", + " 1.0,\n", + " 0.0,\n", + " 0.0,\n", + " 0.0,\n", + " 0.0,\n", + " 0.0\n", + " ],\n", + " [\n", + " 0.0,\n", + " 342.0,\n", + " 5.0,\n", + " 2.0,\n", + " 13.0,\n", + " 2.0,\n", + " 13.0\n", + " ],\n", + " [\n", + " 2.0,\n", + " 10.0,\n", + " 42.0,\n", + " 1.0,\n", + " 12.0,\n", + " 0.0,\n", + " 2.0\n", + " ],\n", + " [\n", + " 0.0,\n", + " 4.0,\n", + " 0.0,\n", + " 121.0,\n", + " 1.0,\n", + " 0.0,\n", + " 4.0\n", + " ],\n", + " [\n", + " 2.0,\n", + " 29.0,\n", + " 3.0,\n", + " 2.0,\n", + " 98.0,\n", + " 0.0,\n", + " 6.0\n", + " ],\n", + " [\n", + " 0.0,\n", + " 3.0,\n", + " 0.0,\n", + " 1.0,\n", + " 0.0,\n", + " 21.0,\n", + " 1.0\n", + " ],\n", + " [\n", + " 0.0,\n", + " 7.0,\n", + " 0.0,\n", + " 1.0,\n", + " 6.0,\n", + " 0.0,\n", + " 409.0\n", + " ]\n", + " ]\n", + " },\n", + " \"logLoss\": 0.08260541,\n", + " \"confidenceMetrics\": [\n", + " {\n", + " \"precision\": 0.14285715,\n", + " \"precisionAt1\": 0.88776374,\n", + " \"recall\": 1.0,\n", + " \"f1ScoreAt1\": 0.88776374,\n", + " \"recallAt1\": 0.88776374,\n", + " \"f1Score\": 0.25\n", + " },\n", + " {\n", + " \"f1Score\": 0.8101266,\n", + " \"recall\": 0.9721519,\n", + " \"precision\": 0.69439423,\n", + " \"confidenceThreshold\": 0.05,\n", + " \"recallAt1\": 0.88776374,\n", + " \"precisionAt1\": 0.88776374,\n", + " \"f1ScoreAt1\": 0.88776374\n", + " },\n", + " \n", + " # REMOVED FOR BREVITY\n", + " \n", + " {\n", + " \"confidenceThreshold\": 1.0,\n", + " \"f1Score\": 0.0033698399,\n", + " \"f1ScoreAt1\": 0.0033698399,\n", + " \"precisionAt1\": 1.0,\n", + " \"precision\": 1.0,\n", + " \"recall\": 0.0016877637,\n", + " \"recallAt1\": 0.0016877637\n", + " }\n", + " ],\n", + " \"auPrc\": 0.95903283\n", + " },\n", + " \"createTime\": \"2021-02-26T06:07:48.967028Z\",\n", + " \"sliceDimensions\": [\n", + " \"annotationSpec\"\n", + " ]\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sIbqsuWeDdiJ" + }, + "source": [ + "## Make batch predictions" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "8USAjugOPmBE" + }, + "source": [ + "### Prepare files for batch prediction" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "IHa-ducKPmBE" + }, + "outputs": [], + "source": [ + "test_item = ! gsutil cat $IMPORT_FILE | head -n1\n", + "test_item, test_label = str(test_item[0]).split(\",\")\n", + "\n", + "print(test_item, test_label)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "y0ThHBwoPmBF" + }, + "source": [ + "*Example output*:\n", + "```\n", + "I went on a successful date with someone I felt sympathy and connection with. affection\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "_BMIZpOXPmBF" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "Let's now make a batch input file, which you store in your local Cloud Storage bucket. The batch input file can be either CSV or JSONL. You will use JSONL in this tutorial. For JSONL file, you make one dictionary entry per line for each text file. The dictionary contains the key/value pairs:\n", + "\n", + "- `content`: The Cloud Storage path to the text file.\n", + "- `mimeType`: The content type. In our example, it is an `text/plain` file.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "4ocY6QmHDdiJ" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "import tensorflow as tf\n", + "\n", + "test_item_uri = \"gs://\" + BUCKET_NAME + \"/test.txt\"\n", + "with tf.io.gfile.GFile(test_item_uri, \"w\") as f:\n", + " f.write(test_item + \"\\n\")\n", + "\n", + "gcs_input_uri = \"gs://\" + BUCKET_NAME + \"/test.jsonl\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " data = {\"content\": test_item_uri, \"mime_type\": \"text/plain\"}\n", + " f.write(json.dumps(data) + \"\\n\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ztl9lP54PmBG" + }, + "outputs": [], + "source": [ + "! gsutil cat $gcs_input_uri\n", + "! gsutil cat $test_item_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pAosq_OoPmBG" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\"content\": \"gs://migration-ucaip-trainingaip-20210226015238/test.txt\", \"mime_type\": \"text/plain\"}\n", + "I went on a successful date with someone I felt sympathy and connection with.\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "YISzKhdUDdiJ" + }, + "source": [ + "### [projects.locations.batchPredictionJobs.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.batchPredictionJobs/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ozaMOwdNJjkT" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "0ZSuOk8PDdiK" + }, + "outputs": [], + "source": [ + "batch_prediction_job = {\n", + " \"display_name\": \"happiness_\" + TIMESTAMP,\n", + " \"model\": model_id,\n", + " \"input_config\": {\n", + " \"instances_format\": \"jsonl\",\n", + " \"gcs_source\": {\"uris\": [gcs_input_uri]},\n", + " },\n", + " \"output_config\": {\n", + " \"predictions_format\": \"jsonl\",\n", + " \"gcs_destination\": {\n", + " \"output_uri_prefix\": \"gs://\" + f\"{BUCKET_NAME}/batch_output/\"\n", + " },\n", + " },\n", + " \"dedicated_resources\": {\n", + " \"machine_spec\": {\n", + " \"machine_type\": \"n1-standard-2\",\n", + " \"accelerator_count\": 0,\n", + " },\n", + " \"starting_replica_count\": 1,\n", + " \"max_replica_count\": 1,\n", + " },\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateBatchPredictionJobRequest(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "swz5uddlPmBH" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"batchPredictionJob\": {\n", + " \"displayName\": \"happiness_20210226015238\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/2369051733671280640\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210226015238/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210226015238/batch_output/\"\n", + " }\n", + " },\n", + " \"dedicatedResources\": {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-2\"\n", + " },\n", + " \"startingReplicaCount\": 1,\n", + " \"maxReplicaCount\": 1\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "vHkuSZxkJjkT" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bvn9xCIZDdiK" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].create_batch_prediction_job(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "FLqkZMD2JjkT" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "PRO-_9tZDdiL" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gXCCM1C5PmBI" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/batchPredictionJobs/4770983263059574784\",\n", + " \"displayName\": \"happiness_20210226015238\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/2369051733671280640\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210226015238/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210226015238/batch_output/\"\n", + " }\n", + " },\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"completionStats\": {\n", + " \"incompleteCount\": \"-1\"\n", + " },\n", + " \"createTime\": \"2021-02-26T09:37:44.471843Z\",\n", + " \"updateTime\": \"2021-02-26T09:37:44.471843Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "-uhh056UPmBI" + }, + "outputs": [], + "source": [ + "# The fully qualified ID for the batch job\n", + "batch_job_id = request.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split(\"/\")[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "5UlgopM2DdiL" + }, + "source": [ + "### [projects.locations.batchPredictionJobs.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.batchPredictionJobs/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ONQ_S59sDdiM" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "TmfmAJcyDdiM" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].get_batch_prediction_job(name=batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sXVZTJyUDdiM" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "eeBVoWoADdiN" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Iqyjp1jBPmBJ" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/batchPredictionJobs/4770983263059574784\",\n", + " \"displayName\": \"happiness_20210226015238\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/2369051733671280640\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210226015238/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210226015238/batch_output/\"\n", + " }\n", + " },\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"completionStats\": {\n", + " \"incompleteCount\": \"-1\"\n", + " },\n", + " \"createTime\": \"2021-02-26T09:37:44.471843Z\",\n", + " \"updateTime\": \"2021-02-26T09:37:44.471843Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "def get_latest_predictions(gcs_out_dir):\n", + " \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n", + " folders = !gsutil ls $gcs_out_dir\n", + " latest = \"\"\n", + " for folder in folders:\n", + " subfolder = folder.split(\"/\")[-2]\n", + " if subfolder.startswith(\"prediction-\"):\n", + " if subfolder > latest:\n", + " latest = folder[:-1]\n", + " return latest\n", + "\n", + "\n", + "while True:\n", + " response = clients[\"job\"].get_batch_prediction_job(name=batch_job_id)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", response.state)\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " folder = get_latest_predictions(\n", + " response.output_config.gcs_destination.output_uri_prefix\n", + " )\n", + " ! gsutil ls $folder/prediction*.jsonl\n", + "\n", + " ! gsutil cat $folder/prediction*.jsonl\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "u3A8vXPAPmBK" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210226015238/batch_output/prediction-happiness_20210226015238-2021-02-26T09:37:44.261133Z/predictions_00001.jsonl\n", + "{\"instance\":{\"content\":\"gs://migration-ucaip-trainingaip-20210226015238/test.txt\",\"mimeType\":\"text/plain\"},\"prediction\":{\"ids\":[\"8446203133482237952\",\"3834517115054850048\",\"1528674105841156096\",\"5563899371965120512\",\"952213353537732608\",\"3258056362751426560\",\"6140360124268544000\"],\"displayNames\":[\"affection\",\"bonding\",\"achievement\",\"enjoy_the_moment\",\"exercise\",\"leisure\",\"nature\"],\"confidences\":[0.9183423,0.045685068,0.024327256,0.0057157497,0.0040851077,0.0012627868,5.8173126E-4]}}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "k0vH7elnDdiN" + }, + "source": [ + "## Make online predictions" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "COwVZtxhJjkW" + }, + "source": [ + "### [projects.locations.endpoints.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "QjWCFUQADdiO" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ZGavOazTDdiO" + }, + "outputs": [], + "source": [ + "endpoint = {\"display_name\": \"happiness_\" + TIMESTAMP}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateEndpointRequest(parent=PARENT, endpoint=endpoint).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "HcwQyxKWPmBM" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"endpoint\": {\n", + " \"displayName\": \"happiness_20210226015238\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "mLlY_jczDdiO" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "_aEAOrzODdiO" + }, + "outputs": [], + "source": [ + "request = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "OG-RyJFeDdiP" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "p1Q2ZqOvDdiP" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Acb1LDrBPmBN" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/endpoints/7367713068517687296\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "KcBsnK1vPmBN" + }, + "outputs": [], + "source": [ + "# The fully qualified ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gpnVJS3AJjkW" + }, + "source": [ + "### [projects.locations.endpoints.deployModel](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/deployModel)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "NuN8rVoMDdiP" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "T8MSMXljDdiP" + }, + "outputs": [], + "source": [ + "deployed_model = {\n", + " \"model\": model_id,\n", + " \"display_name\": \"happiness_\" + TIMESTAMP,\n", + " \"automatic_resources\": {\"min_replica_count\": 1, \"max_replica_count\": 1},\n", + "}\n", + "\n", + "traffic_split = {\"0\": 100}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.DeployModelRequest(\n", + " endpoint=endpoint_id,\n", + " deployed_model=deployed_model,\n", + " traffic_split=traffic_split,\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gwh7ujzbPmBO" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"endpoint\": \"projects/116273516712/locations/us-central1/endpoints/7367713068517687296\",\n", + " \"deployedModel\": {\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/2369051733671280640\",\n", + " \"displayName\": \"happiness_20210226015238\",\n", + " \"automaticResources\": {\n", + " \"minReplicaCount\": 1,\n", + " \"maxReplicaCount\": 1\n", + " }\n", + " },\n", + " \"trafficSplit\": {\n", + " \"0\": 100\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Dh-h_nBZDdiQ" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wZVvs9ixDdiQ" + }, + "outputs": [], + "source": [ + "request = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint_id, deployed_model=deployed_model, traffic_split=traffic_split\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pu8ZfuUGDdiQ" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "o1TLdECDDdiQ" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "5D0-_Md8PmBP" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"deployedModel\": {\n", + " \"id\": \"418518105996656640\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deployed_model_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The unique ID for the deployed model\n", + "deployed_model_id = result.deployed_model.id\n", + "\n", + "print(deployed_model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "I-IV01LtDdiQ" + }, + "source": [ + "### [projects.locations.endpoints.predict](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/predict)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "LcBcyGWEDdiQ" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "LhRgG10QDdiR" + }, + "outputs": [], + "source": [ + "test_item = ! gsutil cat $IMPORT_FILE | head -n1\n", + "test_item, test_label = str(test_item[0]).split(\",\")\n", + "\n", + "instances_list = [{\"content\": test_item}]\n", + "instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + "request = aip.PredictRequest(\n", + " endpoint=endpoint_id,\n", + ")\n", + "request.instances.append(instances)\n", + "\n", + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gpDpf_bqPmBR" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"endpoint\": \"projects/116273516712/locations/us-central1/endpoints/7367713068517687296\",\n", + " \"instances\": [\n", + " [\n", + " {\n", + " \"content\": \"I went on a successful date with someone I felt sympathy and connection with.\"\n", + " }\n", + " ]\n", + " ]\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "B59s9eGxDdiR" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "6QSF5K1ZDdiR" + }, + "outputs": [], + "source": [ + "request = clients[\"prediction\"].predict(endpoint=endpoint_id, instances=instances)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ePJm-GgbDdiR" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "yeGeRM-UDdiR" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "_3qjhEU4PmBS" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"predictions\": [\n", + " {\n", + " \"confidences\": [\n", + " 0.8867673277854919,\n", + " 0.024743923917412758,\n", + " 0.0034913308918476105,\n", + " 0.07936617732048035,\n", + " 0.0013463868526741862,\n", + " 0.0002393187169218436,\n", + " 0.0040455833077430725\n", + " ],\n", + " \"displayNames\": [\n", + " \"affection\",\n", + " \"achievement\",\n", + " \"enjoy_the_moment\",\n", + " \"bonding\",\n", + " \"leisure\",\n", + " \"nature\",\n", + " \"exercise\"\n", + " ],\n", + " \"ids\": [\n", + " \"8446203133482237952\",\n", + " \"1528674105841156096\",\n", + " \"5563899371965120512\",\n", + " \"3834517115054850048\",\n", + " \"3258056362751426560\",\n", + " \"6140360124268544000\",\n", + " \"952213353537732608\"\n", + " ]\n", + " }\n", + " ],\n", + " \"deployedModelId\": \"418518105996656640\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new" + }, + "source": [ + "### [projects.locations.endpoints.undeployModel](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/undeployModel)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "call:migration" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_undeploymodel:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint_id, deployed_model_id=deployed_model_id, traffic_split={}\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "response:migration" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,response" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "21vUY8uePmBU" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bQ-VVaSxJjkd" + }, + "source": [ + "# Cleaning up\r\n", + "\r\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\r\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\r\n", + "\r\n", + "Otherwise, you can delete the individual resources you created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "SyXHVpimDdiS" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_pipeline = True\n", + "delete_batchjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex AI fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset:\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex AI fully qualified identifier for the model\n", + "try:\n", + " if delete_model:\n", + " clients[\"model\"].delete_model(name=model_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex AI fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint:\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex AI fully qualified identifier for the training pipeline\n", + "try:\n", + " if delete_pipeline:\n", + " clients[\"pipeline\"].delete_training_pipeline(name=training_pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex AI fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob:\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r gs://$BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "collapsed_sections": [ + "0yDWDaKwF48e", + "MDUAZaN3JjkL", + "hIHTX-pkJjkO", + "4x_t-MWnJjkQ" + ], + "name": "UJ6 unified AutoML Natural Language Text Classification.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/migration/UJ7 legacy AutoML Natural Language Text Entity Extraction.ipynb b/notebooks/community/migration/UJ7 legacy AutoML Natural Language Text Entity Extraction.ipynb new file mode 100644 index 000000000..e9cd0186e --- /dev/null +++ b/notebooks/community/migration/UJ7 legacy AutoML Natural Language Text Entity Extraction.ipynb @@ -0,0 +1,1633 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "L3g1QPqPjIBe" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_title:migration,automl,icn" + }, + "source": [ + "# AutoML Text entity extractionn model\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "JE-aKjayJjkF" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of AutoML SDK." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "7dxB6-B0JjkG" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-automl" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Iddo-8HmJjkH" + }, + "source": [ + "Install the Google *cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "xHvi6EFSJjkH" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-storage" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ktRTgB8DJjkI" + }, + "source": [ + "### Restart the Kernel\n", + "\n", + "Once you've installed the AutoML SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "x1g4mBdlJjkI" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "if not os.getenv(\"AUTORUN\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "-ppwXGR7ki5K" + }, + "source": [ + "## Before you begin\r\n", + "\r\n", + "### GPU run-time\r\n", + "\r\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\r\n", + "\r\n", + "### Set up your GCP project\r\n", + "\r\n", + "**The following steps are required, regardless of your notebook environment.**\r\n", + "\r\n", + "1. [Select or create a GCP project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\r\n", + "\r\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\r\n", + "\r\n", + "3. [Enable the AutoML APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\r\n", + "\r\n", + "4. [Google Cloud SDK](https://cloud.google.com/sdk) is already installed in AutoML Notebooks.\r\n", + "\r\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\r\n", + "Cloud SDK uses the right project for all the commands in this notebook.\r\n", + "\r\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands.\r\n", + "\r\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "0UfsdLCEJjkJ" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "jT0fijejJjkK" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for AutoML. We recommend when possible, to choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You cannot use a Multi-Regional Storage bucket for training with AutoML. Not all regions provide support for all AutoML services. For the latest support per region, see [Region support for AutoML services]()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "CTBlncfrJjkK" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "J_croPBoJjkK" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "KKxStk5bJjkL" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "AL7KXzUHlwfP" + }, + "source": [ + "### Authenticate your GCP account\r\n", + "\r\n", + "**If you are using AutoML Notebooks**, your environment is already\r\n", + "authenticated. Skip this step.\r\n", + "\r\n", + "*Note: If you are on an AutoML notebook and run the cell, the cell knows to skip executing the authentication steps.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "chybg3Ap_i_2" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your Google Cloud account. This provides access\n", + "# to your Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Vertex, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this tutorial in a notebook locally, replace the string\n", + " # below with the path to your service account key and run this cell to\n", + " # authenticate your Google Cloud account.\n", + " else:\n", + " %env GOOGLE_APPLICATION_CREDENTIALS your_path_to_credentials.json\n", + "\n", + " # Log in to your account on Google Cloud\n", + " ! gcloud auth login" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "MDUAZaN3JjkL" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets. " + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n", + " BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "yU6VuylGJjkM" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "qPPrwWpO_i_6" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Mi-Pdh6QJjkN" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "fn744B7x_i_7" + }, + "outputs": [], + "source": [ + "! gsutil ls -al gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "fR9geV9pJjkO" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "hIHTX-pkJjkO" + }, + "source": [ + "#### Import AutoML SDK\n", + "\n", + "Import the AutoM SDK into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "qrSkIKVgJjkO" + }, + "outputs": [], + "source": [ + "import json\n", + "import os\n", + "import sys\n", + "import time\n", + "\n", + "from google.cloud import automl\n", + "from google.protobuf.json_format import MessageToJson\n", + "from google.protobuf.struct_pb2 import Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "lbv411XjJjkP" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Setup up the following constants for AutoML:\n", + "\n", + "- `PARENT`: The AutoM location root path for dataset, model and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Gx8Eos8PJjkP" + }, + "outputs": [], + "source": [ + "# AutoM location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "l5e_7u3pjIBu" + }, + "source": [ + "## Clients\n", + "\n", + "The AutoML SDK works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the server (AutoML).\n", + "\n", + "You will use several clients in this tutorial, so set them all up upfront.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "c3sR3OxmxXiR" + }, + "outputs": [], + "source": [ + "def automl_client():\n", + " return automl.AutoMlClient()\n", + "\n", + "\n", + "def prediction_client():\n", + " return automl.PredictionServiceClient()\n", + "\n", + "\n", + "def operations_client():\n", + " return automl.AutoMlClient()._transport.operations_client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"automl\"] = automl_client()\n", + "clients[\"prediction\"] = prediction_client()\n", + "clients[\"operations\"] = operations_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:flowers,csv,icn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://cloud-ml-data/NL-entity/dataset.csv\"" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ed725169cbfa" + }, + "outputs": [], + "source": [ + "! gsutil cat $IMPORT_FILE | head -n 10" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bb8bf52f9992" + }, + "source": [ + "*Example output*:\n", + "```\n", + "TRAIN,gs://cloud-ml-data/NL-entity/train.jsonl\n", + "TEST,gs://cloud-ml-data/NL-entity/test.jsonl\n", + "VALIDATION,gs://cloud-ml-data/NL-entity/validation.jsonl\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_create_dataset:migration" + }, + "source": [ + "## Create a dataset" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ChPZpfxLjIBv" + }, + "source": [ + "### Prepare data" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_datasets_create:migration,old" + }, + "source": [ + "### [projects.locations.datasets.create](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.datasets/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ozaMOwdNJjkT" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gJFAw1PejIBw" + }, + "outputs": [], + "source": [ + "dataset = {\n", + " \"display_name\": \"entity_\" + TIMESTAMP,\n", + " \"text_extraction_dataset_metadata\": {},\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " automl.CreateDatasetRequest(parent=PARENT, dataset=dataset).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "60695859c20f" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"dataset\": {\n", + " \"displayName\": \"entity_20210303201139\",\n", + " \"textExtractionDatasetMetadata\": {}\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "vHkuSZxkJjkT" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "zKBpcBR1jIBx" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].create_dataset(parent=PARENT, dataset=dataset)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "FLqkZMD2JjkT" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pOcUKfcZjIBy" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "523c7df9b387" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/datasets/TEN4244124229064196096\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "81ab3645e933" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_datasets_importdata:migration,old" + }, + "source": [ + "### [projects.locations.datasets.importData](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.datasets/importData)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "rwfF7rjvjIBy" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "TI39F1VNjIBy" + }, + "outputs": [], + "source": [ + "input_config = {\"gcs_source\": {\"input_uris\": [IMPORT_FILE]}}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " automl.ImportDataRequest(name=dataset_id, input_config=input_config).__dict__[\n", + " \"_pb\"\n", + " ]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cb5870fde44a" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/datasets/TEN4244124229064196096\",\n", + " \"inputConfig\": {\n", + " \"gcsSource\": {\n", + " \"inputUris\": [\n", + " \"gs://cloud-ml-data/NL-entity/dataset.csv\"\n", + " ]\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "BxnviF7TjIBz" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "HYIPQuJljIBz" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].import_data(name=dataset_id, input_config=input_config)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "BbL7nR-ZjIBz" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "2mBgw1pujIBz" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ce6a7658ebb8" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_create_and_deploy_model:migration" + }, + "source": [ + "## Train a model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_models_create:migration,old" + }, + "source": [ + "### [projects.locations.models.create](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tocf5wHGjIB0" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "rV_FgnILjIB0" + }, + "outputs": [], + "source": [ + "model = {\n", + " \"display_name\": \"entity_\" + TIMESTAMP,\n", + " \"dataset_id\": dataset_short_id,\n", + " \"text_extraction_model_metadata\": {},\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(automl.CreateModelRequest(parent=PARENT, model=model).__dict__[\"_pb\"])\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "71846394320b" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"model\": {\n", + " \"displayName\": \"entity_20210303201139\",\n", + " \"datasetId\": \"TEN4244124229064196096\",\n", + " \"textExtractionModelMetadata\": {}\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "9WGDh5UXjIB1" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Rmu6uXvKjIB1" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].create_model(parent=PARENT, model=model)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "4vQdzcVIjIB2" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "f7c18460f8af" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "c8df6b452c01" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/TEN7821373765161320448\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "adfa99c5e7e5" + }, + "outputs": [], + "source": [ + "# The full unique ID for the training pipeline\n", + "model_id = result.name\n", + "# The short numeric ID for the training pipeline\n", + "model_short_id = model_id.split(\"/\")[-1]\n", + "\n", + "print(model_short_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_evaluate_the_model:migration" + }, + "source": [ + "## Evaluate the model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "yfufMwAEJjkX" + }, + "source": [ + "### [projects.locations.models.modelEvaluations.list](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models.modelEvaluations/list)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "IyTk9SWDjIB3" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tRqkQRB6jIB3" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].list_model_evaluations(parent=model_id, filter=\"\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "YCuTwqG5jIB3" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "AxEY7WFHj2YC" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "model_evaluations = [json.loads(MessageToJson(me.__dict__[\"_pb\"])) for me in request]\n", + "# The evaluation slice\n", + "evaluation_slice = request.model_evaluation[0].name\n", + "\n", + "print(json.dumps(model_evaluations, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "532b930d42e8" + }, + "source": [ + "*Example output*:\n", + "```\n", + "[\n", + " {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/TEN7821373765161320448/modelEvaluations/132746642406774043\",\n", + " \"createTime\": \"2021-03-03T22:30:27.832506Z\",\n", + " \"evaluatedExampleCount\": 60,\n", + " \"textExtractionEvaluationMetrics\": {\n", + " \"confidenceMetricsEntries\": [\n", + " {\n", + " \"confidenceThreshold\": 0.04,\n", + " \"recall\": 0.79928315,\n", + " \"precision\": 0.7950089,\n", + " \"f1Score\": 0.7971403\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.96,\n", + " \"recall\": 0.75089604,\n", + " \"precision\": 0.8603696,\n", + " \"f1Score\": 0.80191386\n", + " },\n", + " \n", + " # REMOVED FOR BREVITY\n", + " {\n", + " \"confidenceThreshold\": 0.43,\n", + " \"recall\": 0.5913978,\n", + " \"precision\": 0.57894737,\n", + " \"f1Score\": 0.5851064\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.44,\n", + " \"recall\": 0.5913978,\n", + " \"precision\": 0.57894737,\n", + " \"f1Score\": 0.5851064\n", + " }\n", + " ]\n", + " },\n", + " \"displayName\": \"DiseaseClass\"\n", + " }\n", + "]\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "i6T0bzuNJjkY" + }, + "source": [ + "### [projects.locations.models.modelEvaluations.get](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models.modelEvaluations/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "1iIWWgs5jIB4" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "jusjAC_tjIB4" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].get_model_evaluation(name=evaluation_slice)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "XFRb-04njIB4" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "oaV7FTA2jIB4" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "490ee6cf06c7" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/TEN7821373765161320448/modelEvaluations/132746642406774043\",\n", + " \"createTime\": \"2021-03-03T22:30:27.832506Z\",\n", + " \"evaluatedExampleCount\": 60,\n", + " \"textExtractionEvaluationMetrics\": {\n", + " \"confidenceMetricsEntries\": [\n", + " {\n", + " \"confidenceThreshold\": 0.04,\n", + " \"recall\": 0.79928315,\n", + " \"precision\": 0.7950089,\n", + " \"f1Score\": 0.7971403\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.96,\n", + " \"recall\": 0.75089604,\n", + " \"precision\": 0.8603696,\n", + " \"f1Score\": 0.80191386\n", + " },\n", + " \n", + " # REMOVED FOR BREVITY\n", + " \n", + " {\n", + " \"confidenceThreshold\": 0.43,\n", + " \"recall\": 0.7921147,\n", + " \"precision\": 0.7935368,\n", + " \"f1Score\": 0.7928251\n", + " },\n", + " {\n", + " \"confidenceThreshold\": 0.44,\n", + " \"recall\": 0.7921147,\n", + " \"precision\": 0.7935368,\n", + " \"f1Score\": 0.7928251\n", + " }\n", + " ]\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "_vTbZbgejIB5" + }, + "source": [ + "## Make batch predictions" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "_NxhzDVNjIB5" + }, + "source": [ + "### Prepare files for batch prediction" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tKzpf6YcjIB5" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "test_item = 'Molecular basis of hexosaminidase A deficiency and pseudodeficiency in the Berks County Pennsylvania Dutch.\\\\tFollowing the birth of two infants with Tay-Sachs disease ( TSD ) , a non-Jewish , Pennsylvania Dutch kindred was screened for TSD carriers using the biochemical assay . A high frequency of individuals who appeared to be TSD heterozygotes was detected ( Kelly et al . , 1975 ) . Clinical and biochemical evidence suggested that the increased carrier frequency was due to at least two altered alleles for the hexosaminidase A alpha-subunit . We now report two mutant alleles in this Pennsylvania Dutch kindred , and one polymorphism . One allele , reported originally in a French TSD patient ( Akli et al . , 1991 ) , is a GT-- > AT transition at the donor splice-site of intron 9 . The second , a C-- > T transition at nucleotide 739 ( Arg247Trp ) , has been shown by Triggs-Raine et al . ( 1992 ) to be a clinically benign \" pseudodeficient \" allele associated with reduced enzyme activity against artificial substrate . Finally , a polymorphism [ G-- > A ( 759 ) ] , which leaves valine at codon 253 unchanged , is described'\n", + "\n", + "gcs_input_uri = \"gs://\" + BUCKET_NAME + \"/test.jsonl\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " data = {\"id\": 0, \"text_snippet\": {\"content\": test_item}}\n", + " f.write(json.dumps(data) + \"\\n\")\n", + "\n", + "! gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "5d84b3d33043" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\"id\": 0, \"text_snippet\": {\"content\": \"Molecular basis of hexosaminidase A deficiency and pseudodeficiency in the Berks County Pennsylvania Dutch.\\\\tFollowing the birth of two infants with Tay-Sachs disease ( TSD ) , a non-Jewish , Pennsylvania Dutch kindred was screened for TSD carriers using the biochemical assay . A high frequency of individuals who appeared to be TSD heterozygotes was detected ( Kelly et al . , 1975 ) . Clinical and biochemical evidence suggested that the increased carrier frequency was due to at least two altered alleles for the hexosaminidase A alpha-subunit . We now report two mutant alleles in this Pennsylvania Dutch kindred , and one polymorphism . One allele , reported originally in a French TSD patient ( Akli et al . , 1991 ) , is a GT-- > AT transition at the donor splice-site of intron 9 . The second , a C-- > T transition at nucleotide 739 ( Arg247Trp ) , has been shown by Triggs-Raine et al . ( 1992 ) to be a clinically benign \\\" pseudodeficient \\\" allele associated with reduced enzyme activity against artificial substrate . Finally , a polymorphism [ G-- > A ( 759 ) ] , which leaves valine at codon 253 unchanged , is described\"}}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_models_batchpredict:migration,old" + }, + "source": [ + "### [projects.locations.models.batchPredict](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models/batchPredict)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "2-oWYpTejIB6" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "EHJqVT8-jIB7" + }, + "outputs": [], + "source": [ + "input_config = {\"gcs_source\": {\"input_uris\": [gcs_input_uri]}}\n", + "\n", + "output_config = {\n", + " \"gcs_destination\": {\"output_uri_prefix\": \"gs://\" + f\"{BUCKET_NAME}/batch_output/\"}\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " automl.BatchPredictRequest(\n", + " name=model_id, input_config=input_config, output_config=output_config\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "d940957b9b51" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/TEN7821373765161320448\",\n", + " \"inputConfig\": {\n", + " \"gcsSource\": {\n", + " \"inputUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210303201139/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"outputConfig\": {\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210303201139/batch_output/\"\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "-iQl862AjIB7" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "XI_kiWLIjIB7" + }, + "outputs": [], + "source": [ + "request = clients[\"prediction\"].batch_predict(\n", + " name=model_id, input_config=input_config, output_config=output_config\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Z9CU4SeXjIB7" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "iGJZl4OCjIB8" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "b6f393652740" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "fnYRCde3xXie" + }, + "outputs": [], + "source": [ + "destination_uri = output_config[\"gcs_destination\"][\"output_uri_prefix\"][:-1]\n", + "\n", + "! gsutil ls $destination_uri/*\n", + "! gsutil cat $destination_uri/prediction*/*.jsonl" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "f8f88ce43155" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210303201139/batch_output/prediction-entity_20210303201139-2021-03-03T22:30:36.292153Z/text_extraction_1.jsonl\n", + "{\"textSnippet\":{\"content\":\"Molecular basis of hexosaminidase A deficiency and pseudodeficiency in the Berks County Pennsylvania Dutch.\\\\tFollowing the birth of two infants with Tay-Sachs disease ( TSD ) , a non-Jewish , Pennsylvania Dutch kindred was screened for TSD carriers using the biochemical assay . A high frequency of individuals who appeared to be TSD heterozygotes was detected ( Kelly et al . , 1975 ) . Clinical and biochemical evidence suggested that the increased carrier frequency was due to at least two altered alleles for the hexosaminidase A alpha-subunit . We now report two mutant alleles in this Pennsylvania Dutch kindred , and one polymorphism . One allele , reported originally in a French TSD patient ( Akli et al . , 1991 ) , is a GT-- \\u003e AT transition at the donor splice-site of intron 9 . The second , a C-- \\u003e T transition at nucleotide 739 ( Arg247Trp ) , has been shown by Triggs-Raine et al . ( 1992 ) to be a clinically benign \\\" pseudodeficient \\\" allele associated with reduced enzyme activity against artificial substrate . Finally , a polymorphism [ G-- \\u003e A ( 759 ) ] , which leaves valine at codon 253 unchanged , is described\"},\"annotations\":[{\"displayName\":\"SpecificDisease\",\"textExtraction\":{\"score\":0.99955064,\"textSegment\":{\"startOffset\":\"19\",\"endOffset\":\"46\",\"content\":\"hexosaminidase A deficiency\"}}},{\"displayName\":\"SpecificDisease\",\"textExtraction\":{\"score\":0.9995449,\"textSegment\":{\"startOffset\":\"149\",\"endOffset\":\"166\",\"content\":\"Tay-Sachs disease\"}}},{\"displayName\":\"SpecificDisease\",\"textExtraction\":{\"score\":0.99939877,\"textSegment\":{\"startOffset\":\"169\",\"endOffset\":\"172\",\"content\":\"TSD\"}}},{\"displayName\":\"Modifier\",\"textExtraction\":{\"score\":0.9993252,\"textSegment\":{\"startOffset\":\"236\",\"endOffset\":\"239\",\"content\":\"TSD\"}}},{\"displayName\":\"Modifier\",\"textExtraction\":{\"score\":0.9993484,\"textSegment\":{\"startOffset\":\"330\",\"endOffset\":\"333\",\"content\":\"TSD\"}}},{\"displayName\":\"Modifier\",\"textExtraction\":{\"score\":0.9993844,\"textSegment\":{\"startOffset\":\"688\",\"endOffset\":\"691\",\"content\":\"TSD\"}}}],\"id\":\"0\"}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "h6qDTq6wjIB8" + }, + "source": [ + "## Make online predictions" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_models_deploy:migration,old" + }, + "source": [ + "### [projects.locations.models.deploy](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models/deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "rCm7yXUHjIB_" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "rlNO4ZgYjIB_" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].deploy_model(name=model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "huJESda4jICA" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "W-h9tZunjICA" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "_E6nl1x7v_NL" + }, + "source": [ + "*Example output*:\r\n", + "```\r\n", + "{}\r\n", + "```" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_models_predict:migration,old" + }, + "source": [ + "### [projects.locations.models.predict](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models/predict)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "2YRTF6AMjIB8" + }, + "source": [ + "#### Prepare data item for online prediction" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "GJaj0apojIB-" + }, + "outputs": [], + "source": [ + "test_item = 'Molecular basis of hexosaminidase A deficiency and pseudodeficiency in the Berks County Pennsylvania Dutch.\\\\tFollowing the birth of two infants with Tay-Sachs disease ( TSD ) , a non-Jewish , Pennsylvania Dutch kindred was screened for TSD carriers using the biochemical assay . A high frequency of individuals who appeared to be TSD heterozygotes was detected ( Kelly et al . , 1975 ) . Clinical and biochemical evidence suggested that the increased carrier frequency was due to at least two altered alleles for the hexosaminidase A alpha-subunit . We now report two mutant alleles in this Pennsylvania Dutch kindred , and one polymorphism . One allele , reported originally in a French TSD patient ( Akli et al . , 1991 ) , is a GT-- > AT transition at the donor splice-site of intron 9 . The second , a C-- > T transition at nucleotide 739 ( Arg247Trp ) , has been shown by Triggs-Raine et al . ( 1992 ) to be a clinically benign \" pseudodeficient \" allele associated with reduced enzyme activity against artificial substrate . Finally , a polymorphism [ G-- > A ( 759 ) ] , which leaves valine at codon 253 unchanged , is described'" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "BM8_JnXbjICA" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "J0cTri_7jICB" + }, + "outputs": [], + "source": [ + "payload = {\"text_snippet\": {\"content\": test_item, \"mime_type\": \"text/plain\"}}\n", + "\n", + "request = automl.PredictRequest(\n", + " name=model_id,\n", + " payload=payload,\n", + ")\n", + "\n", + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bad8702fc564" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/TEN7821373765161320448\",\n", + " \"payload\": {\n", + " \"textSnippet\": {\n", + " \"content\": \"Molecular basis of hexosaminidase A deficiency and pseudodeficiency in the Berks County Pennsylvania Dutch.\\\\tFollowing the birth of two infants with Tay-Sachs disease ( TSD ) , a non-Jewish , Pennsylvania Dutch kindred was screened for TSD carriers using the biochemical assay . A high frequency of individuals who appeared to be TSD heterozygotes was detected ( Kelly et al . , 1975 ) . Clinical and biochemical evidence suggested that the increased carrier frequency was due to at least two altered alleles for the hexosaminidase A alpha-subunit . We now report two mutant alleles in this Pennsylvania Dutch kindred , and one polymorphism . One allele , reported originally in a French TSD patient ( Akli et al . , 1991 ) , is a GT-- > AT transition at the donor splice-site of intron 9 . The second , a C-- > T transition at nucleotide 739 ( Arg247Trp ) , has been shown by Triggs-Raine et al . ( 1992 ) to be a clinically benign \\\" pseudodeficient \\\" allele associated with reduced enzyme activity against artificial substrate . Finally , a polymorphism [ G-- > A ( 759 ) ] , which leaves valine at codon 253 unchanged , is described\",\n", + " \"mimeType\": \"text/plain\"\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "VC6-vTaIjICB" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "PYfH7boujICB" + }, + "outputs": [], + "source": [ + "request = clients[\"prediction\"].predict(request=request)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "e5wP7xa_jICB" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "3tF8D4HejICB" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "1620b1949e82" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"payload\": [\n", + " {\n", + " \"annotationSpecId\": \"8605379431835369472\",\n", + " \"displayName\": \"SpecificDisease\",\n", + " \"textExtraction\": {\n", + " \"score\": 0.99955064,\n", + " \"textSegment\": {\n", + " \"startOffset\": \"19\",\n", + " \"endOffset\": \"46\",\n", + " \"content\": \"hexosaminidase A deficiency\"\n", + " }\n", + " }\n", + " },\n", + " {\n", + " \"annotationSpecId\": \"8605379431835369472\",\n", + " \"displayName\": \"SpecificDisease\",\n", + " \"textExtraction\": {\n", + " \"score\": 0.9995449,\n", + " \"textSegment\": {\n", + " \"startOffset\": \"149\",\n", + " \"endOffset\": \"166\",\n", + " \"content\": \"Tay-Sachs disease\"\n", + " }\n", + " }\n", + " },\n", + " {\n", + " \"annotationSpecId\": \"8605379431835369472\",\n", + " \"displayName\": \"SpecificDisease\",\n", + " \"textExtraction\": {\n", + " \"score\": 0.99939877,\n", + " \"textSegment\": {\n", + " \"startOffset\": \"169\",\n", + " \"endOffset\": \"172\",\n", + " \"content\": \"TSD\"\n", + " }\n", + " }\n", + " },\n", + " {\n", + " \"annotationSpecId\": \"3417232661104558080\",\n", + " \"displayName\": \"Modifier\",\n", + " \"textExtraction\": {\n", + " \"score\": 0.9993252,\n", + " \"textSegment\": {\n", + " \"startOffset\": \"236\",\n", + " \"endOffset\": \"239\",\n", + " \"content\": \"TSD\"\n", + " }\n", + " }\n", + " },\n", + " {\n", + " \"annotationSpecId\": \"3417232661104558080\",\n", + " \"displayName\": \"Modifier\",\n", + " \"textExtraction\": {\n", + " \"score\": 0.9993484,\n", + " \"textSegment\": {\n", + " \"startOffset\": \"330\",\n", + " \"endOffset\": \"333\",\n", + " \"content\": \"TSD\"\n", + " }\n", + " }\n", + " },\n", + " {\n", + " \"annotationSpecId\": \"3417232661104558080\",\n", + " \"displayName\": \"Modifier\",\n", + " \"textExtraction\": {\n", + " \"score\": 0.9993844,\n", + " \"textSegment\": {\n", + " \"startOffset\": \"688\",\n", + " \"endOffset\": \"691\",\n", + " \"content\": \"TSD\"\n", + " }\n", + " }\n", + " }\n", + " ]\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bQ-VVaSxJjkd" + }, + "source": [ + "# Cleaning up\r\n", + "\r\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\r\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\r\n", + "\r\n", + "Otherwise, you can delete the individual resources you created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "qGtpM6SjjICC" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_model = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the AutoML fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset:\n", + " clients[\"automl\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the AutoML fully qualified identifier for the model\n", + "try:\n", + " if delete_model:\n", + " clients[\"automl\"].delete_model(name=model_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r gs://$BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "UJ7 legacy AutoML Natural Language Text Entity Extraction.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/migration/UJ7 unified AutoML Natural Language Text Entity Extraction.ipynb b/notebooks/community/migration/UJ7 unified AutoML Natural Language Text Entity Extraction.ipynb new file mode 100644 index 000000000..48d7e4f64 --- /dev/null +++ b/notebooks/community/migration/UJ7 unified AutoML Natural Language Text Entity Extraction.ipynb @@ -0,0 +1,2417 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Hha2I6DQmDyW" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_title:migration,automl,icn" + }, + "source": [ + "# Vertex AI AutoML text entity extractionn model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "JE-aKjayJjkF" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest (preview) version of Vertex SDK." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "7dxB6-B0JjkG" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-aiplatform --user" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Iddo-8HmJjkH" + }, + "source": [ + "Install the Google *cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "xHvi6EFSJjkH" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-storage" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ktRTgB8DJjkI" + }, + "source": [ + "### Restart the Kernel\n", + "\n", + "Once you've installed the Vertex SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "x1g4mBdlJjkI" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "if not os.getenv(\"AUTORUN\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "3Sq3sGfdt89E" + }, + "source": [ + "## Before you begin\r\n", + "\r\n", + "### GPU run-time\r\n", + "\r\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\r\n", + "\r\n", + "### Set up your GCP project\r\n", + "\r\n", + "**The following steps are required, regardless of your notebook environment.**\r\n", + "\r\n", + "1. [Select or create a GCP project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\r\n", + "\r\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\r\n", + "\r\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\r\n", + "\r\n", + "4. [Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebooks.\r\n", + "\r\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\r\n", + "Cloud SDK uses the right project for all the commands in this notebook.\r\n", + "\r\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "hceMthWGJjkI" + }, + "source": [ + "#### Project ID\n", + "\n", + "**If you don't know your project ID**, try to get your project ID using `gcloud` command by executing the second cell below." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "eidueLnCJjkJ" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "NJ9v1pdtmDyr" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "0UfsdLCEJjkJ" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "jT0fijejJjkK" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex AI. We recommend when possible, to choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You cannot use a Multi-Regional Storage bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see [Region support for Vertex AI services](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "CTBlncfrJjkK" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "J_croPBoJjkK" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "KKxStk5bJjkL" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "9zpjPUOhvRQz" + }, + "source": [ + "### Authenticate your GCP account\r\n", + "\r\n", + "**If you are using Google Cloud Notebooks**, your environment is already\r\n", + "authenticated. Skip this step.\r\n", + "\r\n", + "*Note: If you are on an Vertex notebook and run the cell, the cell knows to skip executing the authentication steps.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "chybg3Ap_i_2" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your Google Cloud account. This provides access\n", + "# to your Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Vertex, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this tutorial in a notebook locally, replace the string\n", + " # below with the path to your service account key and run this cell to\n", + " # authenticate your Google Cloud account.\n", + " else:\n", + " %env GOOGLE_APPLICATION_CREDENTIALS your_path_to_credentials.json\n", + "\n", + " # Log in to your account on Google Cloud\n", + " ! gcloud auth login" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "MDUAZaN3JjkL" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets. " + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n", + " BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "yU6VuylGJjkM" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "QVRixM4oJjkM" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Mi-Pdh6QJjkN" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aqDptKpYJjkN" + }, + "outputs": [], + "source": [ + "! gsutil ls -al gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "fR9geV9pJjkO" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "hIHTX-pkJjkO" + }, + "source": [ + "#### Import Vertex SDK\n", + "\n", + "Import the Vertex SDK into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "qrSkIKVgJjkO" + }, + "outputs": [], + "source": [ + "import json\n", + "import os\n", + "import sys\n", + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "lbv411XjJjkP" + }, + "source": [ + "#### Vertex AI constants\n", + "\n", + "Setup up the following constants for Vertex AI:\n", + "\n", + "- `API_ENDPOINT`: The Vertex AI API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex AI location root path for dataset, model and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Gx8Eos8PJjkP" + }, + "outputs": [], + "source": [ + "# API Endpoint\n", + "API_ENDPOINT = \"us-central1-aiplatform.googleapis.com\"\n", + "\n", + "# Vertex AI location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "4x_t-MWnJjkQ" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Next, setup constants unique to AutoML Text Entity Extraction datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the managed dataset service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the managed dataset service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the managed pipelines service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "nmzhH42tJjkQ" + }, + "outputs": [], + "source": [ + "# Text Dataset type\n", + "TEXT_SCHEMA = \"google-cloud-aiplatform/schema/dataset/metadata/text_1.0.0.yaml\"\n", + "# Text Labeling type\n", + "IMPORT_SCHEMA_TEXT_EXTRACTION = \"gs://google-cloud-aiplatform/schema/dataset/ioformat/text_extraction_io_format_1.0.0.yaml\"\n", + "# Text Training task\n", + "TRAINING_TEXT_EXTRACTION_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_text_extraction_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "R7P3MRNgJjkQ" + }, + "source": [ + "## Clients\n", + "\n", + "The Vertex SDK works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the server (Vertex).\n", + "\n", + "You will use several clients in this tutorial, so set them all up upfront.\n", + "\n", + "- Dataset Service for managed datasets.\n", + "- Model Service for managed models.\n", + "- Pipeline Service for training.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving. *Note*: Prediction has a different service endpoint." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "kt8j_ey4JjkR" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "clients[\"job\"] = create_job_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:flowers,csv,icn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://cloud-samples-data/language/ucaip_ten_dataset.jsonl\"" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cBgFrDD1_i__" + }, + "outputs": [], + "source": [ + "! gsutil cat $IMPORT_FILE | head -n 1" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_create:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{'text_segment_annotations': [{'endOffset': 54, 'startOffset': 27, 'displayName': 'SpecificDisease'}, {'endOffset': 173, 'startOffset': 156, 'displayName': 'SpecificDisease'}, {'endOffset': 179, 'startOffset': 176, 'displayName': 'SpecificDisease'}, {'endOffset': 246, 'startOffset': 243, 'displayName': 'Modifier'}, {'endOffset': 340, 'startOffset': 337, 'displayName': 'Modifier'}, {'endOffset': 698, 'startOffset': 695, 'displayName': 'Modifier'}], 'textContent': '1301937\\tMolecular basis of hexosaminidase A deficiency and pseudodeficiency in the Berks County Pennsylvania Dutch.\\tFollowing the birth of two infants with Tay-Sachs disease ( TSD ) , a non-Jewish , Pennsylvania Dutch kindred was screened for TSD carriers using the biochemical assay . A high frequency of individuals who appeared to be TSD heterozygotes was detected ( Kelly et al . , 1975 ) . Clinical and biochemical evidence suggested that the increased carrier frequency was due to at least two altered alleles for the hexosaminidase A alpha-subunit . We now report two mutant alleles in this Pennsylvania Dutch kindred , and one polymorphism . One allele , reported originally in a French TSD patient ( Akli et al . , 1991 ) , is a GT-- > AT transition at the donor splice-site of intron 9 . The second , a C-- > T transition at nucleotide 739 ( Arg247Trp ) , has been shown by Triggs-Raine et al . ( 1992 ) to be a clinically benign \" pseudodeficient \" allele associated with reduced enzyme activity against artificial substrate . Finally , a polymorphism [ G-- > A ( 759 ) ] , which leaves valine at codon 253 unchanged , is described .\\n '}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_create_dataset:migration" + }, + "source": [ + "## Create a dataset" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_datasets_create:migration,new" + }, + "source": [ + "### [projects.locations.datasets.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.datasets/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "trXpXKsKmDy2" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "LieciZovmDy2" + }, + "outputs": [], + "source": [ + "DATA_SCHEMA = TEXT_SCHEMA\n", + "\n", + "dataset = {\n", + " \"display_name\": \"ten_\" + TIMESTAMP,\n", + " \"metadata_schema_uri\": \"gs://\" + DATA_SCHEMA,\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateDatasetRequest(parent=PARENT, dataset=dataset).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"dataset\": {\n", + " \"displayName\": \"ten_20210301154552\",\n", + " \"metadataSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/metadata/text_1.0.0.yaml\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "-CD22Lm9mDy2" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ptyxODU2mDy3" + }, + "outputs": [], + "source": [ + "request = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "vpUO9HBimDy3" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "jvKS4F-QmDy3" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/datasets/1309228077611483136\",\n", + " \"displayName\": \"ten_20210301154552\",\n", + " \"metadataSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/metadata/text_1.0.0.yaml\",\n", + " \"labels\": {\n", + " \"aiplatform.googleapis.com/dataset_metadata_schema\": \"TEXT\"\n", + " },\n", + " \"metadata\": {\n", + " \"dataItemSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/dataitem/text_1.0.0.yaml\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_datasets_import:migration,new" + }, + "source": [ + "### [projects.locations.datasets.import](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.datasets/import)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "amH_XuOjmDy4" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "5_5_ZY-pmDy4" + }, + "outputs": [], + "source": [ + "LABEL_SCHEMA = IMPORT_SCHEMA_TEXT_EXTRACTION\n", + "\n", + "import_config = {\n", + " \"gcs_source\": {\"uris\": [IMPORT_FILE]},\n", + " \"import_schema_uri\": LABEL_SCHEMA,\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.ImportDataRequest(name=dataset_id, import_configs=[import_config]).__dict__[\n", + " \"_pb\"\n", + " ]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/datasets/1309228077611483136\",\n", + " \"importConfigs\": [\n", + " {\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://cloud-samples-data/language/ucaip_ten_dataset.jsonl\"\n", + " ]\n", + " },\n", + " \"importSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/ioformat/text_extraction_io_format_1.0.0.yaml\"\n", + " }\n", + " ]\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "w52itF0ymDy4" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "4hxz5c70mDy4" + }, + "outputs": [], + "source": [ + "request = clients[\"dataset\"].import_data(\n", + " name=dataset_id, import_configs=[import_config]\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "oSDxPS2TmDy4" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "jCW-M3tgmDy5" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_create_and_deploy_model:migration" + }, + "source": [ + "## Train a model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0oqIBOSnJjkW" + }, + "source": [ + "### [projects.locations.trainingPipelines.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.trainingPipelines/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "t9wJ2d03mDy5" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tJf08imCmDy5" + }, + "outputs": [], + "source": [ + "TRAINING_SCHEMA = TRAINING_TEXT_EXTRACTION_SCHEMA\n", + "\n", + "task = Value(\n", + " struct_value=Struct(\n", + " fields={\n", + " \"multi_label\": Value(bool_value=False),\n", + " \"budget_milli_node_hours\": Value(number_value=1000),\n", + " \"model_type\": Value(string_value=\"CLOUD\"),\n", + " \"disable_early_stopping\": Value(bool_value=False),\n", + " }\n", + " )\n", + ")\n", + "\n", + "training_pipeline = {\n", + " \"display_name\": \"ten_\" + TIMESTAMP,\n", + " \"input_data_config\": {\"dataset_id\": dataset_short_id},\n", + " \"model_to_upload\": {\"display_name\": \"ten_\" + TIMESTAMP},\n", + " \"training_task_definition\": TRAINING_SCHEMA,\n", + " \"training_task_inputs\": task,\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateTrainingPipelineRequest(\n", + " parent=PARENT,\n", + " training_pipeline=training_pipeline,\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"trainingPipeline\": {\n", + " \"displayName\": \"ten_20210301154552\",\n", + " \"inputDataConfig\": {\n", + " \"datasetId\": \"1309228077611483136\"\n", + " },\n", + " \"trainingTaskDefinition\": \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_text_extraction_1.0.0.yaml\",\n", + " \"trainingTaskInputs\": {\n", + " \"budget_milli_node_hours\": 1000.0,\n", + " \"multi_label\": false,\n", + " \"model_type\": \"CLOUD\",\n", + " \"disable_early_stopping\": false\n", + " },\n", + " \"modelToUpload\": {\n", + " \"displayName\": \"ten_20210301154552\"\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "yk19A8OKmDy6" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "OOyUwmq_mDy6" + }, + "outputs": [], + "source": [ + "request = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "_W-OpMk7mDy6" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "5z4U8eJSmDy6" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/trainingPipelines/4643220011912003584\",\n", + " \"displayName\": \"ten_20210301154552\",\n", + " \"inputDataConfig\": {\n", + " \"datasetId\": \"1309228077611483136\"\n", + " },\n", + " \"trainingTaskDefinition\": \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_text_extraction_1.0.0.yaml\",\n", + " \"trainingTaskInputs\": {},\n", + " \"modelToUpload\": {\n", + " \"displayName\": \"ten_20210301154552\"\n", + " },\n", + " \"state\": \"PIPELINE_STATE_PENDING\",\n", + " \"createTime\": \"2021-03-01T15:55:29.352065Z\",\n", + " \"updateTime\": \"2021-03-01T15:55:29.352065Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "u8sLc67Pf7L9" + }, + "outputs": [], + "source": [ + "# The full unique ID for the training pipeline\n", + "training_pipeline_id = request.name\n", + "# The short numeric ID for the training pipeline\n", + "training_pipeline_short_id = training_pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(training_pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "QJsV1xoSJjkW" + }, + "source": [ + "### [projects.locations.trainingPipelines.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.trainingPipelines/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "m40wuXKgmDy6" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "yp_mYeyrmDy7" + }, + "outputs": [], + "source": [ + "request = clients[\"pipeline\"].get_training_pipeline(name=training_pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "j_-_Ue1WmDy7" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "2xbLIG-8mDy7" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/trainingPipelines/4643220011912003584\",\n", + " \"displayName\": \"ten_20210301154552\",\n", + " \"inputDataConfig\": {\n", + " \"datasetId\": \"1309228077611483136\"\n", + " },\n", + " \"trainingTaskDefinition\": \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_text_extraction_1.0.0.yaml\",\n", + " \"trainingTaskInputs\": {},\n", + " \"modelToUpload\": {\n", + " \"displayName\": \"ten_20210301154552\"\n", + " },\n", + " \"state\": \"PIPELINE_STATE_PENDING\",\n", + " \"createTime\": \"2021-03-01T15:55:29.352065Z\",\n", + " \"updateTime\": \"2021-03-01T15:55:29.352065Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=training_pipeline_id)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " break\n", + " else:\n", + " model_id = response.model_to_upload.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "print(model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_evaluate_the_model:migration" + }, + "source": [ + "## Evaluate the model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "BJNzhELXJjkY" + }, + "source": [ + "### [projects.locations.models.evaluations.list](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.models.evaluations/list)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "c9k8GnUemDy7" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cESW6zutmDy8" + }, + "outputs": [], + "source": [ + "request = clients[\"model\"].list_model_evaluations(parent=model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bGFL_p9SmDy8" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "prEvwExMmDy8" + }, + "outputs": [], + "source": [ + "model_evaluations = [json.loads(MessageToJson(mel.__dict__[\"_pb\"])) for mel in request]\n", + "\n", + "# The evaluation slice\n", + "evaluation_slice = request.model_evaluations[0].name\n", + "\n", + "print(json.dumps(model_evaluations, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "[\n", + " {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/4400738115568795648/evaluations/7959028222912364544\",\n", + " \"metricsSchemaUri\": \"gs://google-cloud-aiplatform/schema/modelevaluation/text_extraction_metrics_1.0.0.yaml\",\n", + " \"metrics\": {\n", + " \"confusionMatrix\": {\n", + " \"rows\": [\n", + " [\n", + " 0.0,\n", + " 24.0,\n", + " 23.0,\n", + " 1.0,\n", + " 27.0\n", + " ],\n", + " [\n", + " 9.0,\n", + " 40.0,\n", + " 0.0,\n", + " 0.0,\n", + " 10.0\n", + " ],\n", + " [\n", + " 11.0,\n", + " 0.0,\n", + " 87.0,\n", + " 0.0,\n", + " 2.0\n", + " ],\n", + " [\n", + " 3.0,\n", + " 0.0,\n", + " 0.0,\n", + " 5.0,\n", + " 0.0\n", + " ],\n", + " [\n", + " 32.0,\n", + " 16.0,\n", + " 7.0,\n", + " 1.0,\n", + " 186.0\n", + " ]\n", + " ],\n", + " \"annotationSpecs\": [\n", + " {\n", + " \"displayName\": \"NULL\"\n", + " },\n", + " {\n", + " \"id\": \"2041829376663748608\",\n", + " \"displayName\": \"DiseaseClass\"\n", + " },\n", + " {\n", + " \"displayName\": \"Modifier\",\n", + " \"id\": \"4347672385877442560\"\n", + " },\n", + " {\n", + " \"displayName\": \"CompositeMention\",\n", + " \"id\": \"6653515395091136512\"\n", + " },\n", + " {\n", + " \"id\": \"7806436899697983488\",\n", + " \"displayName\": \"SpecificDisease\"\n", + " }\n", + " ]\n", + " },\n", + " \"confidenceMetrics\": [\n", + " {\n", + " \"precision\": 0.74125874,\n", + " \"f1Score\": 0.7589499,\n", + " \"recall\": 0.7775061,\n", + " \"confidenceThreshold\": 0.04\n", + " },\n", + " {\n", + " \"recall\": 0.7457213,\n", + " \"confidenceThreshold\": 0.96,\n", + " \"precision\": 0.8333333,\n", + " \"f1Score\": 0.7870968\n", + " },\n", + " \n", + " # REMOVED FOR BREVITY\n", + " \n", + " {\n", + " \"f1Score\": 0.7596154,\n", + " \"recall\": 0.77261615,\n", + " \"confidenceThreshold\": 0.44,\n", + " \"precision\": 0.7470449\n", + " }\n", + " ]\n", + " },\n", + " \"createTime\": \"2021-03-01T17:59:23.638307Z\",\n", + " \"sliceDimensions\": [\n", + " \"annotationSpec\"\n", + " ]\n", + " }\n", + "]\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wr2JuU5nJjka" + }, + "source": [ + "### [projects.locations.models.evaluations.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.models.evaluations/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "7DKMBMcEmDy8" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "CXakTdZ7mDy8" + }, + "outputs": [], + "source": [ + "request = clients[\"model\"].get_model_evaluation(name=evaluation_slice)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "_Ch_w1a9mDy9" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "LKXnVatsmDy9" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/4400738115568795648/evaluations/7959028222912364544\",\n", + " \"metricsSchemaUri\": \"gs://google-cloud-aiplatform/schema/modelevaluation/text_extraction_metrics_1.0.0.yaml\",\n", + " \"metrics\": {\n", + " \"confusionMatrix\": {\n", + " \"rows\": [\n", + " [\n", + " 0.0,\n", + " 24.0,\n", + " 23.0,\n", + " 1.0,\n", + " 27.0\n", + " ],\n", + " [\n", + " 9.0,\n", + " 40.0,\n", + " 0.0,\n", + " 0.0,\n", + " 10.0\n", + " ],\n", + " [\n", + " 11.0,\n", + " 0.0,\n", + " 87.0,\n", + " 0.0,\n", + " 2.0\n", + " ],\n", + " [\n", + " 3.0,\n", + " 0.0,\n", + " 0.0,\n", + " 5.0,\n", + " 0.0\n", + " ],\n", + " [\n", + " 32.0,\n", + " 16.0,\n", + " 7.0,\n", + " 1.0,\n", + " 186.0\n", + " ]\n", + " ],\n", + " \"annotationSpecs\": [\n", + " {\n", + " \"displayName\": \"NULL\"\n", + " },\n", + " {\n", + " \"id\": \"2041829376663748608\",\n", + " \"displayName\": \"DiseaseClass\"\n", + " },\n", + " {\n", + " \"displayName\": \"Modifier\",\n", + " \"id\": \"4347672385877442560\"\n", + " },\n", + " {\n", + " \"id\": \"6653515395091136512\",\n", + " \"displayName\": \"CompositeMention\"\n", + " },\n", + " {\n", + " \"displayName\": \"SpecificDisease\",\n", + " \"id\": \"7806436899697983488\"\n", + " }\n", + " ]\n", + " },\n", + " \"confidenceMetrics\": [\n", + " {\n", + " \"precision\": 0.74125874,\n", + " \"recall\": 0.7775061,\n", + " \"confidenceThreshold\": 0.04,\n", + " \"f1Score\": 0.7589499\n", + " },\n", + " {\n", + " \"f1Score\": 0.7870968,\n", + " \"recall\": 0.7457213,\n", + " \"confidenceThreshold\": 0.96,\n", + " \"precision\": 0.8333333\n", + " },\n", + " \n", + " # REMOVED FOR BREVITY\n", + " \n", + " {\n", + " \"precision\": 0.745283,\n", + " \"f1Score\": 0.7587035,\n", + " \"recall\": 0.77261615,\n", + " \"confidenceThreshold\": 0.43\n", + " },\n", + " {\n", + " \"precision\": 0.7470449,\n", + " \"recall\": 0.77261615,\n", + " \"confidenceThreshold\": 0.44,\n", + " \"f1Score\": 0.7596154\n", + " }\n", + " ]\n", + " },\n", + " \"createTime\": \"2021-03-01T17:59:23.638307Z\",\n", + " \"sliceDimensions\": [\n", + " \"annotationSpec\"\n", + " ]\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "OnCI5OoPmDy9" + }, + "source": [ + "## Make batch predictions" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gM-YixlLmDy9" + }, + "source": [ + "### Make a batch prediction file\r\n", + "\r\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "V6caf56xmDy9" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "import tensorflow as tf\n", + "\n", + "test_item = 'Molecular basis of hexosaminidase A deficiency and pseudodeficiency in the Berks County Pennsylvania Dutch.\\\\tFollowing the birth of two infants with Tay-Sachs disease ( TSD ) , a non-Jewish , Pennsylvania Dutch kindred was screened for TSD carriers using the biochemical assay . A high frequency of individuals who appeared to be TSD heterozygotes was detected ( Kelly et al . , 1975 ) . Clinical and biochemical evidence suggested that the increased carrier frequency was due to at least two altered alleles for the hexosaminidase A alpha-subunit . We now report two mutant alleles in this Pennsylvania Dutch kindred , and one polymorphism . One allele , reported originally in a French TSD patient ( Akli et al . , 1991 ) , is a GT-- > AT transition at the donor splice-site of intron 9 . The second , a C-- > T transition at nucleotide 739 ( Arg247Trp ) , has been shown by Triggs-Raine et al . ( 1992 ) to be a clinically benign \" pseudodeficient \" allele associated with reduced enzyme activity against artificial substrate . Finally , a polymorphism [ G-- > A ( 759 ) ] , which leaves valine at codon 253 unchanged , is described'\n", + "\n", + "gcs_test_item = \"gs://\" + BUCKET_NAME + \"/test.txt\"\n", + "with tf.io.gfile.GFile(gcs_test_item, \"w\") as f:\n", + " f.write(test_item + \"\\n\")\n", + "\n", + "gcs_input_uri = \"gs://\" + BUCKET_NAME + \"/test.jsonl\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " f.write(json.dumps({\"content\": gcs_test_item, \"mime_type\": \"text/plain\"}) + \"\\n\")\n", + "\n", + "! gsutil cat $gcs_input_uri\n", + "! gsutil cat $gcs_test_item" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\"content\": \"gs://migration-ucaip-trainingaip-20210301154552/test.txt\", \"mime_type\": \"text/plain\"}\n", + "Molecular basis of hexosaminidase A deficiency and pseudodeficiency in the Berks County Pennsylvania Dutch.\\tFollowing the birth of two infants with Tay-Sachs disease ( TSD ) , a non-Jewish , Pennsylvania Dutch kindred was screened for TSD carriers using the biochemical assay . A high frequency of individuals who appeared to be TSD heterozygotes was detected ( Kelly et al . , 1975 ) . Clinical and biochemical evidence suggested that the increased carrier frequency was due to at least two altered alleles for the hexosaminidase A alpha-subunit . We now report two mutant alleles in this Pennsylvania Dutch kindred , and one polymorphism . One allele , reported originally in a French TSD patient ( Akli et al . , 1991 ) , is a GT-- > AT transition at the donor splice-site of intron 9 . The second , a C-- > T transition at nucleotide 739 ( Arg247Trp ) , has been shown by Triggs-Raine et al . ( 1992 ) to be a clinically benign \" pseudodeficient \" allele associated with reduced enzyme activity against artificial substrate . Finally , a polymorphism [ G-- > A ( 759 ) ] , which leaves valine at codon 253 unchanged , is described\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "5YZGIxwFmDy9" + }, + "source": [ + "### [projects.locations.batchPredictionJobs.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.batchPredictionJobs/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ozaMOwdNJjkT" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "RXRw6E0XmDy-" + }, + "outputs": [], + "source": [ + "batch_prediction_job = {\n", + " \"display_name\": \"ten_\" + TIMESTAMP,\n", + " \"model\": model_id,\n", + " \"input_config\": {\n", + " \"instances_format\": \"jsonl\",\n", + " \"gcs_source\": {\"uris\": [gcs_input_uri]},\n", + " },\n", + " \"output_config\": {\n", + " \"predictions_format\": \"jsonl\",\n", + " \"gcs_destination\": {\n", + " \"output_uri_prefix\": \"gs://\" + f\"{BUCKET_NAME}/batch_output/\"\n", + " },\n", + " },\n", + " \"dedicated_resources\": {\n", + " \"machine_spec\": {\"machine_type\": \"n1-standard-2\", \"accelerator_count\": 0},\n", + " \"starting_replica_count\": 1,\n", + " \"max_replica_count\": 1,\n", + " },\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateBatchPredictionJobRequest(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"batchPredictionJob\": {\n", + " \"displayName\": \"ten_20210301154552\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/4400738115568795648\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210301154552/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210301154552/batch_output/\"\n", + " }\n", + " },\n", + " \"dedicatedResources\": {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-2\"\n", + " },\n", + " \"startingReplicaCount\": 1,\n", + " \"maxReplicaCount\": 1\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sjVy8V2vmDy-" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Gl-oZ4iEmDy-" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].create_batch_prediction_job(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "D8ZPK4CJmDy-" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "5SFADp4amDy-" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/batchPredictionJobs/3588251799200464896\",\n", + " \"displayName\": \"ten_20210301154552\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/4400738115568795648\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210301154552/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210301154552/batch_output/\"\n", + " }\n", + " },\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"completionStats\": {\n", + " \"incompleteCount\": \"-1\"\n", + " },\n", + " \"createTime\": \"2021-03-01T17:59:42.777083Z\",\n", + " \"updateTime\": \"2021-03-01T17:59:42.777083Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "948796f1aecd" + }, + "outputs": [], + "source": [ + "# The fully qualified ID for the batch job\n", + "batch_job_id = request.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split(\"/\")[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "AOgLC0hnmDy_" + }, + "source": [ + "### [projects.locations.batchPredictionJobs.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.batchPredictionJobs/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "YRHUe9-ZmDy_" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "qi6-JkammDy_" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].get_batch_prediction_job(name=batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "KQUdbrjfmDy_" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "h3DRpflrmDy_" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/batchPredictionJobs/3588251799200464896\",\n", + " \"displayName\": \"ten_20210301154552\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/4400738115568795648\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210301154552/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210301154552/batch_output/\"\n", + " }\n", + " },\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"completionStats\": {\n", + " \"incompleteCount\": \"-1\"\n", + " },\n", + " \"createTime\": \"2021-03-01T17:59:42.777083Z\",\n", + " \"updateTime\": \"2021-03-01T17:59:42.777083Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = clients[\"job\"].get_batch_prediction_job(name=batch_job_id)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", response.state)\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " folder = response.output_config.gcs_destination.output_uri_prefix[:-1]\n", + " ! gsutil ls $folder/prediction*/*.jsonl\n", + "\n", + " ! gsutil cat $folder/prediction*/*.jsonl\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "trainingpipelines_create:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210301154552/batch_output/prediction-ten_20210301154552-2021-03-01T17:59:42.638222Z/predictions_00001.jsonl\n", + "{\"instance\":{\"content\":\"gs://migration-ucaip-trainingaip-20210301154552/test.txt\",\"mimeType\":\"text/plain\"},\"prediction\":{\"ids\":[\"7806436899697983488\",\"7806436899697983488\",\"7806436899697983488\",\"4347672385877442560\",\"4347672385877442560\",\"4347672385877442560\"],\"displayNames\":[\"SpecificDisease\",\"SpecificDisease\",\"SpecificDisease\",\"Modifier\",\"Modifier\",\"Modifier\"],\"textSegmentStartOffsets\":[\"149\",\"19\",\"169\",\"236\",\"688\",\"330\"],\"textSegmentEndOffsets\":[\"165\",\"45\",\"171\",\"238\",\"690\",\"332\"],\"confidences\":[0.99957836,0.9995628,0.9995044,0.9993287,0.9993144,0.99927235]}}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wOZktSibmDzA" + }, + "source": [ + "## Make online predictions" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "COwVZtxhJjkW" + }, + "source": [ + "### [projects.locations.endpoints.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "UqsMvafmmDzA" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "WJXSR0SemDzA" + }, + "outputs": [], + "source": [ + "endpoint = {\"display_name\": \"ten_\" + TIMESTAMP}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateEndpointRequest(parent=PARENT, endpoint=endpoint).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"endpoint\": {\n", + " \"displayName\": \"ten_20210301154552\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "TkObt-D9mDzB" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "4yS1e4-umDzB" + }, + "outputs": [], + "source": [ + "request = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "XhlKSNtLmDzB" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "rx_183PQmDzB" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/endpoints/8916247652891361280\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "c36b9ed788b2" + }, + "outputs": [], + "source": [ + "# The fully qualified ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gpnVJS3AJjkW" + }, + "source": [ + "### [projects.locations.endpoints.deployModel](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/deployModel)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "BtJfuq3FmDzC" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "BiHGmlWCmDzC" + }, + "outputs": [], + "source": [ + "deployed_model = {\n", + " \"model\": model_id,\n", + " \"display_name\": \"ten_\" + TIMESTAMP,\n", + " \"automatic_resources\": {\"min_replica_count\": 1, \"max_replica_count\": 1},\n", + "}\n", + "\n", + "traffic_split = {\"0\": 100}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.DeployModelRequest(\n", + " endpoint=endpoint_id,\n", + " deployed_model=deployed_model,\n", + " traffic_split=traffic_split,\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"endpoint\": \"projects/116273516712/locations/us-central1/endpoints/8916247652891361280\",\n", + " \"deployedModel\": {\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/4400738115568795648\",\n", + " \"displayName\": \"ten_20210301154552\",\n", + " \"automaticResources\": {\n", + " \"minReplicaCount\": 1,\n", + " \"maxReplicaCount\": 1\n", + " }\n", + " },\n", + " \"trafficSplit\": {\n", + " \"0\": 100\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "hIzbdl0VmDzC" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "4N0ab1LbmDzC" + }, + "outputs": [], + "source": [ + "request = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint_id, deployed_model=deployed_model, traffic_split=traffic_split\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "3VEwSDuDmDzC" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "usGWKp2OmDzD" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"deployedModel\": {\n", + " \"id\": \"3958065938133155840\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deployed_model_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The unique ID for the deployed model\n", + "deployed_model_id = result.deployed_model.id\n", + "\n", + "print(deployed_model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Q4d0u9hLmDzD" + }, + "source": [ + "### [projects.locations.endpoints.predict](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/predict)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "e1de98a2fd17" + }, + "source": [ + "#### Prepare data item for online prediction" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "kJ5XyBVrmDzA" + }, + "outputs": [], + "source": [ + "test_item = 'Molecular basis of hexosaminidase A deficiency and pseudodeficiency in the Berks County Pennsylvania Dutch.\\\\tFollowing the birth of two infants with Tay-Sachs disease ( TSD ) , a non-Jewish , Pennsylvania Dutch kindred was screened for TSD carriers using the biochemical assay . A high frequency of individuals who appeared to be TSD heterozygotes was detected ( Kelly et al . , 1975 ) . Clinical and biochemical evidence suggested that the increased carrier frequency was due to at least two altered alleles for the hexosaminidase A alpha-subunit . We now report two mutant alleles in this Pennsylvania Dutch kindred , and one polymorphism . One allele , reported originally in a French TSD patient ( Akli et al . , 1991 ) , is a GT-- > AT transition at the donor splice-site of intron 9 . The second , a C-- > T transition at nucleotide 739 ( Arg247Trp ) , has been shown by Triggs-Raine et al . ( 1992 ) to be a clinically benign \" pseudodeficient \" allele associated with reduced enzyme activity against artificial substrate . Finally , a polymorphism [ G-- > A ( 759 ) ] , which leaves valine at codon 253 unchanged , is described'" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Fwp-_YXHmDzD" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "nxF4dCLcmDzD" + }, + "outputs": [], + "source": [ + "instances_list = [{\"content\": test_item}]\n", + "\n", + "instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + "prediction_request = aip.PredictRequest(\n", + " endpoint=endpoint_id,\n", + ")\n", + "prediction_request.instances.append(instances)\n", + "\n", + "print(MessageToJson(prediction_request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"endpoint\": \"projects/116273516712/locations/us-central1/endpoints/8916247652891361280\",\n", + " \"instances\": [\n", + " [\n", + " {\n", + " \"content\": \"Molecular basis of hexosaminidase A deficiency and pseudodeficiency in the Berks County Pennsylvania Dutch.\\\\tFollowing the birth of two infants with Tay-Sachs disease ( TSD ) , a non-Jewish , Pennsylvania Dutch kindred was screened for TSD carriers using the biochemical assay . A high frequency of individuals who appeared to be TSD heterozygotes was detected ( Kelly et al . , 1975 ) . Clinical and biochemical evidence suggested that the increased carrier frequency was due to at least two altered alleles for the hexosaminidase A alpha-subunit . We now report two mutant alleles in this Pennsylvania Dutch kindred , and one polymorphism . One allele , reported originally in a French TSD patient ( Akli et al . , 1991 ) , is a GT-- > AT transition at the donor splice-site of intron 9 . The second , a C-- > T transition at nucleotide 739 ( Arg247Trp ) , has been shown by Triggs-Raine et al . ( 1992 ) to be a clinically benign \\\" pseudodeficient \\\" allele associated with reduced enzyme activity against artificial substrate . Finally , a polymorphism [ G-- > A ( 759 ) ] , which leaves valine at codon 253 unchanged , is described\"\n", + " }\n", + " ]\n", + " ]\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "vHkuSZxkJjkT" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "MmtZNBAHmDzE" + }, + "outputs": [], + "source": [ + "request = clients[\"prediction\"].predict(endpoint=endpoint_id, instances=instances)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "FLqkZMD2JjkT" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "vGx0eDC0mDzE" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"predictions\": [\n", + " {\n", + " \"displayNames\": [\n", + " \"SpecificDisease\",\n", + " \"SpecificDisease\",\n", + " \"SpecificDisease\",\n", + " \"Modifier\",\n", + " \"Modifier\",\n", + " \"Modifier\"\n", + " ],\n", + " \"confidences\": [\n", + " 0.9995627999305725,\n", + " 0.9995783567428589,\n", + " 0.9995043873786926,\n", + " 0.9993286728858948,\n", + " 0.999272346496582,\n", + " 0.9993144273757935\n", + " ],\n", + " \"textSegmentStartOffsets\": [\n", + " 19.0,\n", + " 149.0,\n", + " 169.0,\n", + " 236.0,\n", + " 330.0,\n", + " 688.0\n", + " ],\n", + " \"ids\": [\n", + " \"7806436899697983488\",\n", + " \"7806436899697983488\",\n", + " \"7806436899697983488\",\n", + " \"4347672385877442560\",\n", + " \"4347672385877442560\",\n", + " \"4347672385877442560\"\n", + " ],\n", + " \"textSegmentEndOffsets\": [\n", + " 46.0,\n", + " 166.0,\n", + " 172.0,\n", + " 239.0,\n", + " 333.0,\n", + " 691.0\n", + " ]\n", + " }\n", + " ],\n", + " \"deployedModelId\": \"3958065938133155840\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new" + }, + "source": [ + "### [projects.locations.endpoints.undeployModel](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/undeployModel)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "KrVZz6Uw_jAp" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_undeploymodel:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint_id,\n", + " deployed_model_id=deployed_model_id,\n", + " traffic_split={},\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wvFK-kir_jAq" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "GvajSw-Y_jAq" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bQ-VVaSxJjkd" + }, + "source": [ + "# Cleaning up\r\n", + "\r\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\r\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\r\n", + "\r\n", + "Otherwise, you can delete the individual resources you created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "5SCM9v_zmDzF" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_pipeline = True\n", + "delete_batchjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex AI fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset:\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex AI fully qualified identifier for the model\n", + "try:\n", + " if delete_model:\n", + " clients[\"model\"].delete_model(name=model_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex AI fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint:\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex AI fully qualified identifier for the training pipeline\n", + "try:\n", + " if delete_pipeline:\n", + " clients[\"pipeline\"].delete_training_pipeline(name=training_pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex AI fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob:\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r gs://$BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "UJ7 unified AutoML Natural Language Text Entity Extraction.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/migration/UJ8 legacy AutoML Natural Language - Text Sentiment Analysis.ipynb b/notebooks/community/migration/UJ8 legacy AutoML Natural Language - Text Sentiment Analysis.ipynb new file mode 100644 index 000000000..982b8b3d2 --- /dev/null +++ b/notebooks/community/migration/UJ8 legacy AutoML Natural Language - Text Sentiment Analysis.ipynb @@ -0,0 +1,1634 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Xo4cz5r0c4yy" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_title:migration,automl,icn" + }, + "source": [ + "# AutoML text sentiment analysis\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "JE-aKjayJjkF" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of AutoML SDK." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "7dxB6-B0JjkG" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-automl" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Iddo-8HmJjkH" + }, + "source": [ + "Install the Google *cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "xHvi6EFSJjkH" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-storage" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ktRTgB8DJjkI" + }, + "source": [ + "### Restart the Kernel\n", + "\n", + "Once you've installed the AutoML SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "x1g4mBdlJjkI" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "if not os.getenv(\"AUTORUN\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "AGX8yszodHx_" + }, + "source": [ + "## Before you begin\r\n", + "\r\n", + "### GPU run-time\r\n", + "\r\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\r\n", + "\r\n", + "### Set up your GCP project\r\n", + "\r\n", + "**The following steps are required, regardless of your notebook environment.**\r\n", + "\r\n", + "1. [Select or create a GCP project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\r\n", + "\r\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\r\n", + "\r\n", + "3. [Enable the AutoML APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\r\n", + "\r\n", + "4. [Google Cloud SDK](https://cloud.google.com/sdk) is already installed in AutoML Notebooks.\r\n", + "\r\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\r\n", + "Cloud SDK uses the right project for all the commands in this notebook.\r\n", + "\r\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands.\r\n", + "\r\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "jT0fijejJjkK" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for AutoML. We recommend when possible, to choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You cannot use a Multi-Regional Storage bucket for training with AutoML. Not all regions provide support for all AutoML services. For the latest support per region, see [Region support for AutoML services]()\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "CTBlncfrJjkK" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "J_croPBoJjkK" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "KKxStk5bJjkL" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "c3yKEDsvdpFf" + }, + "source": [ + "### Authenticate your GCP account\r\n", + "\r\n", + "**If you are using AutoML Notebooks**, your environment is already\r\n", + "authenticated. Skip this step.\r\n", + "\r\n", + "*Note: If you are on an AutoML notebook and run the cell, the cell knows to skip executing the authentication steps.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "chybg3Ap_i_2" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your Google Cloud account. This provides access\n", + "# to your Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Vertex, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this tutorial in a notebook locally, replace the string\n", + " # below with the path to your service account key and run this cell to\n", + " # authenticate your Google Cloud account.\n", + " else:\n", + " %env GOOGLE_APPLICATION_CREDENTIALS your_path_to_credentials.json\n", + "\n", + " # Log in to your account on Google Cloud\n", + " ! gcloud auth login" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "MDUAZaN3JjkL" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n", + " BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "yU6VuylGJjkM" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "QVRixM4oJjkM" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Mi-Pdh6QJjkN" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aqDptKpYJjkN" + }, + "outputs": [], + "source": [ + "! gsutil ls -al gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "fR9geV9pJjkO" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "hIHTX-pkJjkO" + }, + "source": [ + "#### Import AutoML SDK\n", + "\n", + "Import the AutoM SDK into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "qrSkIKVgJjkO" + }, + "outputs": [], + "source": [ + "import json\n", + "import os\n", + "import sys\n", + "import time\n", + "\n", + "from google.cloud import automl\n", + "from google.protobuf.json_format import MessageToJson\n", + "from google.protobuf.struct_pb2 import Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "lbv411XjJjkP" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Setup up the following constants for AutoML:\n", + "\n", + "- `PARENT`: The AutoM location root path for dataset, model and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Gx8Eos8PJjkP" + }, + "outputs": [], + "source": [ + "# AutoM location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "hBaOc_HRc4zC" + }, + "source": [ + "## Clients\n", + "\n", + "The AutoML SDK works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the server (AutoML).\n", + "\n", + "You will use several clients in this tutorial, so set them all up upfront." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "5t0zkxqPc4zC" + }, + "outputs": [], + "source": [ + "def automl_client():\n", + " return automl.AutoMlClient()\n", + "\n", + "\n", + "def perdictions_client():\n", + " return automl.PredictionServiceClient()\n", + "\n", + "\n", + "def operations_client():\n", + " return automl.AutoMlClient()._transport.operations_client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"automl\"] = automl_client()\n", + "clients[\"predictions\"] = perdictions_client()\n", + "clients[\"operations\"] = operations_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:flowers,csv,icn" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "IMPORT_FILE = \"gs://cloud-samples-data/language/claritin.csv\"\n", + "with tf.io.gfile.GFile(IMPORT_FILE, \"r\") as f:\n", + " content = f.readlines()\n", + "\n", + "IMPORT_FILE = \"gs://\" + BUCKET_NAME + \"/claritin.csv\"\n", + "with tf.io.gfile.GFile(IMPORT_FILE, \"w\") as f:\n", + " for line in content:\n", + " f.write(\",\".join(line.split(\",\")[0:-1]) + \"\\n\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ed725169cbfa" + }, + "outputs": [], + "source": [ + "! gsutil cat $IMPORT_FILE | head -n 10" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "d3a0464dd6f3" + }, + "source": [ + "*Example output*:\n", + "```\n", + "@freewrytin God is way too good for Claritin,2\n", + "I need Claritin. So bad. When did I become cursed with allergies?,3\n", + "Thank god for Claritin.,4\n", + "\"And what's worse is that I reached my 3-day limit on the nose spray yesterday, which means I have to rely on Claritin.\",2\n", + "Time to take some Claritin or Allegra or something. I need my voice,3\n", + "Oh my RT @imsydneycharles: I just want it to be on record somewhere that I took Claritin and Benadryl together...just in case I pass out,2\n", + "Bouta take a Claritin _‰Ûª‰Û_‰Ûª_‰ÛªÌâ FML !!,3\n", + "Commander Loratadine Generic A Sarcelles: Commander Loratadine Generic A Sarcelles Claritin =‰Ûª_‰Ûª__ http://t.co/mOleL8AM,2\n", + "\"Zyrtec, Claritin, Suddafed, Nasal Spray.. I feel like a drug addict taking these Allergy medicine. Please Allergy season.. DISAPPEAR!!\",1\n", + "\"‰Ûª_‰Ûª_‰ÛªÕ@SheLovesThatD: If she has allergies, give her the Claritin D.‰Ûª_‰Ûª_Ì_å @Sweeno_thakid41 @B_Original16 @luke_CYwalker14\",3\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_create_dataset:migration" + }, + "source": [ + "## Create a dataset" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_datasets_create:migration,old" + }, + "source": [ + "### [projects.locations.datasets.create](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.datasets/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ozaMOwdNJjkT" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "OMZr1t8mc4zF" + }, + "outputs": [], + "source": [ + "dataset = {\n", + " \"display_name\": \"claritin_\" + TIMESTAMP,\n", + " \"text_sentiment_dataset_metadata\": {\"sentiment_max\": 4},\n", + "}\n", + "\n", + "\n", + "print(\n", + " MessageToJson(\n", + " automl.CreateDatasetRequest(parent=PARENT, dataset=dataset).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "f1354773b89e" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"dataset\": {\n", + " \"displayName\": \"claritin_20210304132912\",\n", + " \"textSentimentDatasetMetadata\": {\n", + " \"sentimentMax\": 4\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "vHkuSZxkJjkT" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ifCavN75c4zG" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].create_dataset(parent=PARENT, dataset=dataset)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "FLqkZMD2JjkT" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "87xXwkaVc4zG" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "f9c171ffceee" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/datasets/TST1994716952680988672\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "86e845326428" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_datasets_importdata:migration,old" + }, + "source": [ + "### [projects.locations.datasets.importData](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.datasets/importData)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "y3T7l1xAc4zH" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "jjOtdiJBc4zH" + }, + "outputs": [], + "source": [ + "input_config = {\"gcs_source\": {\"input_uris\": [IMPORT_FILE]}}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " automl.ImportDataRequest(name=dataset_id, input_config=input_config).__dict__[\n", + " \"_pb\"\n", + " ]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "c0fd2b918d71" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/datasets/TST1994716952680988672\",\n", + " \"inputConfig\": {\n", + " \"gcsSource\": {\n", + " \"inputUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210304132912/claritin.csv\"\n", + " ]\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "7QSqElzPc4zH" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "FWm339Twc4zI" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].import_data(name=dataset_id, input_config=input_config)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "AQVw0rQsc4zI" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Fur_ANcMc4zI" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "NT2ras8rhC_D" + }, + "source": [ + "*Example output*:\r\n", + "```\r\n", + "{}\r\n", + "```" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_create_and_deploy_model:migration" + }, + "source": [ + "## Train a model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_models_create:migration,old" + }, + "source": [ + "### [projects.locations.models.create](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bfPBtMYZc4zJ" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "LbWL7tq2c4zJ" + }, + "outputs": [], + "source": [ + "model = {\n", + " \"display_name\": \"claritin_\" + TIMESTAMP,\n", + " \"dataset_id\": dataset_short_id,\n", + " \"text_sentiment_model_metadata\": {},\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(automl.CreateModelRequest(parent=PARENT, model=model).__dict__[\"_pb\"])\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "e080c4730d86" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"model\": {\n", + " \"displayName\": \"claritin_20210304132912\",\n", + " \"datasetId\": \"TST1994716952680988672\",\n", + " \"textSentimentModelMetadata\": {}\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "duJfz52Rc4zJ" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "kHS2pwcec4zK" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].create_model(parent=PARENT, model=model)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "vmkA2d5-c4zK" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "o8P7a8nQc4zK" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "498f8fd21640" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/TST4078882474816438272\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "865fcdfa7262" + }, + "outputs": [], + "source": [ + "# The full unique ID for the training pipeline\n", + "model_id = result.name\n", + "# The short numeric ID for the training pipeline\n", + "model_short_id = model_id.split(\"/\")[-1]\n", + "\n", + "print(model_short_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_evaluate_the_model:migration" + }, + "source": [ + "## Evaluate the model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "yfufMwAEJjkX" + }, + "source": [ + "### [projects.locations.models.modelEvaluations.list](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models.modelEvaluations/list)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ro1Hr7Hbc4zL" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "hBEu4Ghoc4zL" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].list_model_evaluations(parent=model_id, filter=\"\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "X9XycwTYc4zM" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ZbX6Z00uc4zM" + }, + "outputs": [], + "source": [ + "model_evaluations = [json.loads(MessageToJson(me.__dict__[\"_pb\"])) for me in request]\n", + "# The evaluation slice\n", + "evaluation_slice = request.model_evaluation[0].name\n", + "\n", + "print(json.dumps(model_evaluations, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "6835b78da85b" + }, + "source": [ + "*Example output*:\n", + "```\n", + "[\n", + " {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/TST4078882474816438272/modelEvaluations/54870628009945864\",\n", + " \"annotationSpecId\": \"8301667931964571648\",\n", + " \"createTime\": \"2021-03-04T17:15:51.851420Z\",\n", + " \"textSentimentEvaluationMetrics\": {\n", + " \"precision\": 0.33333334,\n", + " \"recall\": 0.16666667,\n", + " \"f1Score\": 0.22222222\n", + " },\n", + " \"displayName\": \"4\"\n", + " },\n", + " {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/TST4078882474816438272/modelEvaluations/1597159550285093673\",\n", + " \"annotationSpecId\": \"1384138904323489792\",\n", + " \"createTime\": \"2021-03-04T17:15:51.851420Z\",\n", + " \"textSentimentEvaluationMetrics\": {\n", + " \"precision\": 0.5,\n", + " \"recall\": 0.296875,\n", + " \"f1Score\": 0.37254903\n", + " },\n", + " \"displayName\": \"1\"\n", + " },\n", + " {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/TST4078882474816438272/modelEvaluations/3521790980763365687\",\n", + " \"createTime\": \"2021-03-04T17:15:51.851420Z\",\n", + " \"evaluatedExampleCount\": 452,\n", + " \"textSentimentEvaluationMetrics\": {\n", + " \"precision\": 0.6238938,\n", + " \"recall\": 0.6238938,\n", + " \"f1Score\": 0.6238938,\n", + " \"meanAbsoluteError\": 0.47566372,\n", + " \"meanSquaredError\": 0.69690263,\n", + " \"linearKappa\": 0.41007927,\n", + " \"quadraticKappa\": 0.45938763,\n", + " \"confusionMatrix\": {\n", + " \"annotationSpecId\": [\n", + " \"7148746427357724672\",\n", + " \"1384138904323489792\",\n", + " \"5995824922750877696\",\n", + " \"3689981913537183744\",\n", + " \"8301667931964571648\"\n", + " ],\n", + " \"row\": [\n", + " {\n", + " \"exampleCount\": [\n", + " 2,\n", + " 4,\n", + " 1,\n", + " 1,\n", + " 1\n", + " ]\n", + " },\n", + " {\n", + " \"exampleCount\": [\n", + " 3,\n", + " 19,\n", + " 14,\n", + " 28,\n", + " 0\n", + " ]\n", + " },\n", + " {\n", + " \"exampleCount\": [\n", + " 0,\n", + " 7,\n", + " 67,\n", + " 63,\n", + " 1\n", + " ]\n", + " },\n", + " {\n", + " \"exampleCount\": [\n", + " 1,\n", + " 8,\n", + " 19,\n", + " 191,\n", + " 4\n", + " ]\n", + " },\n", + " {\n", + " \"exampleCount\": [\n", + " 0,\n", + " 0,\n", + " 0,\n", + " 15,\n", + " 3\n", + " ]\n", + " }\n", + " ],\n", + " \"displayName\": [\n", + " \"0\",\n", + " \"1\",\n", + " \"2\",\n", + " \"3\",\n", + " \"4\"\n", + " ]\n", + " }\n", + " }\n", + " },\n", + " {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/TST4078882474816438272/modelEvaluations/3727703410992997127\",\n", + " \"annotationSpecId\": \"3689981913537183744\",\n", + " \"createTime\": \"2021-03-04T17:15:51.851420Z\",\n", + " \"textSentimentEvaluationMetrics\": {\n", + " \"precision\": 0.6409396,\n", + " \"recall\": 0.85650223,\n", + " \"f1Score\": 0.7332054\n", + " },\n", + " \"displayName\": \"3\"\n", + " },\n", + " {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/TST4078882474816438272/modelEvaluations/4692810493650008310\",\n", + " \"annotationSpecId\": \"7148746427357724672\",\n", + " \"createTime\": \"2021-03-04T17:15:51.851420Z\",\n", + " \"textSentimentEvaluationMetrics\": {\n", + " \"precision\": 0.33333334,\n", + " \"recall\": 0.22222222,\n", + " \"f1Score\": 0.26666668\n", + " },\n", + " \"displayName\": \"0\"\n", + " },\n", + " {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/TST4078882474816438272/modelEvaluations/8390011688796741170\",\n", + " \"annotationSpecId\": \"5995824922750877696\",\n", + " \"createTime\": \"2021-03-04T17:15:51.851420Z\",\n", + " \"textSentimentEvaluationMetrics\": {\n", + " \"precision\": 0.6633663,\n", + " \"recall\": 0.48550725,\n", + " \"f1Score\": 0.5606694\n", + " },\n", + " \"displayName\": \"2\"\n", + " }\n", + "]\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "i6T0bzuNJjkY" + }, + "source": [ + "### [projects.locations.models.modelEvaluations.get](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models.modelEvaluations/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cu1lRCSOc4zM" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "XIIHbDYWc4zM" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].get_model_evaluation(name=evaluation_slice)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aMgYNNWgc4zN" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "hSWpi4v1c4zN" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "b11081c8c71e" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/TST4078882474816438272/modelEvaluations/54870628009945864\",\n", + " \"annotationSpecId\": \"8301667931964571648\",\n", + " \"createTime\": \"2021-03-04T17:15:51.851420Z\",\n", + " \"textSentimentEvaluationMetrics\": {\n", + " \"precision\": 0.33333334,\n", + " \"recall\": 0.16666667,\n", + " \"f1Score\": 0.22222222\n", + " },\n", + " \"displayName\": \"4\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "iRUvxxAKc4zN" + }, + "source": [ + "## Make batch predictions" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "caA77NMbc4zN" + }, + "source": [ + "### Make the batch input file\r\n", + "\r\n", + "To request a batch of predictions from AutoML Video, create a CSV file that lists the Cloud Storage paths to the videos that you want to annotate. You can also specify a start and end time to tell AutoML Video to only annotate a segment (segment-level) of the video. The start time must be zero or greater and must be before the end time. The end time must be greater than the start time and less than or equal to the duration of the video. You can also use inf to indicate the end of a video." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "5beaf1d4077c" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "gcs_input_uri = \"gs://\" + BUCKET_NAME + \"/test.csv\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " item_1 = \"gs://cloud-samples-data/language/sentiment-positive.txt\"\n", + " ! gsutil cp $item_1 gs://$BUCKET_NAME\n", + " f.write(\"gs://\" + BUCKET_NAME + \"/sentiment-positive.txt\" + \"\\n\")\n", + "\n", + " item_2 = \"gs://cloud-samples-data/language/sentiment-negative.txt\"\n", + " ! gsutil cp $item_2 gs://$BUCKET_NAME\n", + " f.write(\"gs://\" + BUCKET_NAME + \"/sentiment-negative.txt\")\n", + "\n", + "! gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "031c110f4c79" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210304132912/sentiment-positive.txt\n", + "gs://migration-ucaip-trainingaip-20210304132912/sentiment-negative.txt\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_models_batchpredict:migration,old" + }, + "source": [ + "### [projects.locations.models.batchPredict](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models/batchPredict)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "UNxo4fR5c4zO" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Vu-8O5Y2c4zO" + }, + "outputs": [], + "source": [ + "input_config = {\"gcs_source\": {\"input_uris\": [gcs_input_uri]}}\n", + "\n", + "output_config = {\n", + " \"gcs_destination\": {\"output_uri_prefix\": \"gs://\" + f\"{BUCKET_NAME}/batch_output/\"}\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " automl.BatchPredictRequest(\n", + " name=model_id, input_config=input_config, output_config=output_config\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "9defd1c4d1a0" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/TST4078882474816438272\",\n", + " \"inputConfig\": {\n", + " \"gcsSource\": {\n", + " \"inputUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210304132912/test.csv\"\n", + " ]\n", + " }\n", + " },\n", + " \"outputConfig\": {\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210304132912/batch_output/\"\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "kc_59mp8c4zP" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "TSk7yRiqc4zP" + }, + "outputs": [], + "source": [ + "request = clients[\"predictions\"].batch_predict(\n", + " name=model_id, input_config=input_config, output_config=output_config\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "mcqKVj2Xc4zP" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bSTEmLgGc4zP" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "c4CvqTBhtMNi" + }, + "source": [ + "*Example output*:\r\n", + "```\r\n", + "{}\r\n", + "```" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "FQjooSKyc4zQ" + }, + "source": [ + "## Make online predictions\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "nrRQoOCVc4zQ" + }, + "source": [ + "#### Prepare data item for online prediction\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "mMdW9VbJc4zQ" + }, + "outputs": [], + "source": [ + "test_data = ! gsutil cat $IMPORT_FILE | head -n1\n", + "\n", + "test_item = str(test_data[0]).split(\",\")[0]\n", + "test_label = str(test_data[0]).split(\",\")[1]\n", + "\n", + "print((test_item, test_label))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "8981b397c864" + }, + "source": [ + "*Example output*:\n", + "```\n", + "('@freewrytin God is way too good for Claritin', '2')\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_models_deploy:migration,old" + }, + "source": [ + "### [projects.locations.models.deploy](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models/deploy)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "h0AxsMSjc4zQ" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "_Uxf6m8Cc4zR" + }, + "outputs": [], + "source": [ + "request = clients[\"automl\"].deploy_model(name=model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Rj8jg_GVc4zR" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "q7rox1YCc4zR" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "s1ikNfXniG-3" + }, + "source": [ + "*Example output*:\r\n", + "```\r\n", + "{}\r\n", + "```" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_models_predict:migration,old" + }, + "source": [ + "### [projects.locations.models.predict](https://cloud.google.com/automl/docs/reference/rest/v1beta1/projects.locations.models/predict)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "g688Vva3c4zS" + }, + "outputs": [], + "source": [ + "payload = {\"text_snippet\": {\"content\": test_item, \"mime_type\": \"text/plain\"}}\n", + "\n", + "prediction_request = automl.PredictRequest(\n", + " name=model_id,\n", + " payload=payload,\n", + ")\n", + "\n", + "print(MessageToJson(prediction_request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cd025e2420db" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/TST4078882474816438272\",\n", + " \"payload\": {\n", + " \"textSnippet\": {\n", + " \"content\": \"@freewrytin God is way too good for Claritin\",\n", + " \"mimeType\": \"text/plain\"\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "kgznWoQcc4zS" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "GUOkxQwic4zS" + }, + "outputs": [], + "source": [ + "request = clients[\"predictions\"].predict(request=prediction_request)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "9MfWbNzhc4zS" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "xZZliEjoc4zS" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dc047af97666" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"payload\": [\n", + " {\n", + " \"textSentiment\": {\n", + " \"sentiment\": 3\n", + " }\n", + " }\n", + " ],\n", + " \"metadata\": {\n", + " \"sentiment_score\": \"0.30955505\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bQ-VVaSxJjkd" + }, + "source": [ + "# Cleaning up\r\n", + "\r\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\r\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\r\n", + "\r\n", + "Otherwise, you can delete the individual resources you created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "oNJs94pKc4zT" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_model = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the AutoML fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset:\n", + " clients[\"automl\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the AutoML fully qualified identifier for the model\n", + "try:\n", + " if delete_model:\n", + " clients[\"automl\"].delete_model(name=model_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r gs://$BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "UJ8 legacy AutoML Natural Language - Text Sentiment Analysis.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/migration/UJ8 unified AutoML Natural Language - Text Sentiment Analysis.ipynb b/notebooks/community/migration/UJ8 unified AutoML Natural Language - Text Sentiment Analysis.ipynb new file mode 100644 index 000000000..ad3a46f69 --- /dev/null +++ b/notebooks/community/migration/UJ8 unified AutoML Natural Language - Text Sentiment Analysis.ipynb @@ -0,0 +1,2355 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Sf2Z7mlw1siQ" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_title:migration,automl,icn" + }, + "source": [ + "# Vertex AI AutoML text sentiment analysis model\n", + "\n", + "## Installation\n", + "\n", + "Install the latest (preview) version of Vertex SDK." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "7dxB6-B0JjkG" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-aiplatform --user" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Iddo-8HmJjkH" + }, + "source": [ + "Install the Google *cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "xHvi6EFSJjkH" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-storage" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ktRTgB8DJjkI" + }, + "source": [ + "### Restart the Kernel\n", + "\n", + "Once you've installed the Vertex SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "x1g4mBdlJjkI" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "if not os.getenv(\"AUTORUN\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "4yZl1w-T6H1Q" + }, + "source": [ + "## Before you begin\r\n", + "\r\n", + "### GPU run-time\r\n", + "\r\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\r\n", + "\r\n", + "### Set up your GCP project\r\n", + "\r\n", + "**The following steps are required, regardless of your notebook environment.**\r\n", + "\r\n", + "1. [Select or create a GCP project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\r\n", + "\r\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\r\n", + "\r\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\r\n", + "\r\n", + "4. [Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebooks.\r\n", + "\r\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\r\n", + "Cloud SDK uses the right project for all the commands in this notebook.\r\n", + "\r\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "eidueLnCJjkJ" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "s__D9HcZ1siZ" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "0UfsdLCEJjkJ" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "jT0fijejJjkK" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex AI. We recommend when possible, to choose the region closest to you. \n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You cannot use a Multi-Regional Storage bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see [Region support for Vertex AI services](https://cloud.google.com/vertex-ai/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "CTBlncfrJjkK" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "J_croPBoJjkK" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "KKxStk5bJjkL" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Wx4DX8I07d-_" + }, + "source": [ + "### Authenticate your GCP account\r\n", + "\r\n", + "**If you are using Google Cloud Notebooks**, your environment is already\r\n", + "authenticated. Skip this step.\r\n", + "\r\n", + "*Note: If you are on an Vertex notebook and run the cell, the cell knows to skip executing the authentication steps.*" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "chybg3Ap_i_2" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your Google Cloud account. This provides access\n", + "# to your Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Vertex, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this tutorial in a notebook locally, replace the string\n", + " # below with the path to your service account key and run this cell to\n", + " # authenticate your Google Cloud account.\n", + " else:\n", + " %env GOOGLE_APPLICATION_CREDENTIALS your_path_to_credentials.json\n", + "\n", + " # Log in to your account on Google Cloud\n", + " ! gcloud auth login" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "MDUAZaN3JjkL" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n", + " BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "yU6VuylGJjkM" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "QVRixM4oJjkM" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Mi-Pdh6QJjkN" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aqDptKpYJjkN" + }, + "outputs": [], + "source": [ + "! gsutil ls -al gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "fR9geV9pJjkO" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "hIHTX-pkJjkO" + }, + "source": [ + "#### Import Vertex SDK\n", + "\n", + "Import the Vertex SDK into our Python environment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "qrSkIKVgJjkO" + }, + "outputs": [], + "source": [ + "import json\n", + "import os\n", + "import sys\n", + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf import json_format\n", + "from google.protobuf.json_format import MessageToJson\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "lbv411XjJjkP" + }, + "source": [ + "#### Vertex AI constants\n", + "\n", + "Setup up the following constants for Vertex AI:\n", + "\n", + "- `API_ENDPOINT`: The Vertex AI API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex AI location root path for dataset, model and endpoint resources." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Gx8Eos8PJjkP" + }, + "outputs": [], + "source": [ + "# API Endpoint\n", + "API_ENDPOINT = \"us-central1-aiplatform.googleapis.com\"\n", + "\n", + "# Vertex AI location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "4x_t-MWnJjkQ" + }, + "source": [ + "#### AutoML constants\n", + "\n", + "Next, setup constants unique to AutoML Text Entity Extraction datasets and training:\n", + "\n", + "- Dataset Schemas: Tells the managed dataset service which type of dataset it is.\n", + "- Data Labeling (Annotations) Schemas: Tells the managed dataset service how the data is labeled (annotated).\n", + "- Dataset Training Schemas: Tells the managed pipelines service the task (e.g., classification) to train the model for." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "nmzhH42tJjkQ" + }, + "outputs": [], + "source": [ + "# Text Dataset type\n", + "TEXT_SCHEMA = \"google-cloud-aiplatform/schema/dataset/metadata/text_1.0.0.yaml\"\n", + "# Text Labeling type\n", + "IMPORT_SCHEMA_TEXT_SENTIMENT = \"gs://google-cloud-aiplatform/schema/dataset/ioformat/text_sentiment_io_format_1.0.0.yaml\"\n", + "# Text Training task\n", + "TRAINING_TEXT_SENTIMENT_SCHEMA = \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_text_sentiment_1.0.0.yaml\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "R7P3MRNgJjkQ" + }, + "source": [ + "## Clients\n", + "\n", + "The Vertex SDK works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the server (Vertex).\n", + "\n", + "You will use several clients in this tutorial, so set them all up upfront.\n", + "\n", + "- Dataset Service for managed datasets.\n", + "- Model Service for managed models.\n", + "- Pipeline Service for training.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving. *Note*: Prediction has a different service endpoint." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "kt8j_ey4JjkR" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_dataset_client():\n", + " client = aip.DatasetServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_pipeline_client():\n", + " client = aip.PipelineServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"dataset\"] = create_dataset_client()\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"pipeline\"] = create_pipeline_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "clients[\"job\"] = create_job_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:flowers,csv,icn" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = \"gs://cloud-samples-data/language/claritin-split.csv\"" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cBgFrDD1_i__" + }, + "outputs": [], + "source": [ + "! gsutil cat $IMPORT_FILE | head -n 10" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_create:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "TRAINING,@freewrytin God is way too good for Claritin,2,4\n", + "TRAINING,I need Claritin. So bad. When did I become cursed with allergies?,3,4\n", + "TRAINING,Thank god for Claritin.,4,4\n", + "TRAINING,\"And what's worse is that I reached my 3-day limit on the nose spray yesterday, which means I have to rely on Claritin.\",2,4\n", + "TRAINING,Time to take some Claritin or Allegra or something. I need my voice,3,4\n", + "TRAINING,Oh my RT @imsydneycharles: I just want it to be on record somewhere that I took Claritin and Benadryl together...just in case I pass out,2,4\n", + "TRAINING,Bouta take a Claritin _‰Ûª‰Û_‰Ûª_‰ÛªÌâ FML !!,3,4\n", + "TRAINING,Commander Loratadine Generic A Sarcelles: Commander Loratadine Generic A Sarcelles Claritin =‰Ûª_‰Ûª__ http://t.co/mOleL8AM,2,4\n", + "TRAINING,\"Zyrtec, Claritin, Suddafed, Nasal Spray.. I feel like a drug addict taking these Allergy medicine. Please Allergy season.. DISAPPEAR!!\",1,4\n", + "TRAINING,\"‰Ûª_‰Ûª_‰ÛªÕ@SheLovesThatD: If she has allergies, give her the Claritin D.‰Ûª_‰Ûª_Ì_å @Sweeno_thakid41 @B_Original16 @luke_CYwalker14\",3,4\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_create_dataset:migration" + }, + "source": [ + "## Create a dataset" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "b0J_eRSR1sih" + }, + "source": [ + "### Prepare data" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_datasets_create:migration,new" + }, + "source": [ + "### [projects.locations.datasets.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.datasets/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "UwqJ-Ua-1sii" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "BsQTgqv41sij" + }, + "outputs": [], + "source": [ + "DATA_SCHEMA = TEXT_SCHEMA\n", + "\n", + "dataset = {\n", + " \"display_name\": \"claritin_\" + TIMESTAMP,\n", + " \"metadata_schema_uri\": \"gs://\" + DATA_SCHEMA,\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateDatasetRequest(parent=PARENT, dataset=dataset).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"dataset\": {\n", + " \"displayName\": \"claritin_20210301212135\",\n", + " \"metadataSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/metadata/text_1.0.0.yaml\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "8Vw-Ht8R1sij" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ctso28tq1sij" + }, + "outputs": [], + "source": [ + "request = clients[\"dataset\"].create_dataset(parent=PARENT, dataset=dataset)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "t2VmcwXf1sij" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "7X0tEzLW1sik" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/datasets/3047617533776494592\",\n", + " \"displayName\": \"claritin_20210301212135\",\n", + " \"metadataSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/metadata/text_1.0.0.yaml\",\n", + " \"labels\": {\n", + " \"aiplatform.googleapis.com/dataset_metadata_schema\": \"TEXT\"\n", + " },\n", + " \"metadata\": {\n", + " \"dataItemSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/dataitem/text_1.0.0.yaml\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dataset_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the dataset\n", + "dataset_id = result.name\n", + "# The short numeric ID for the dataset\n", + "dataset_short_id = dataset_id.split(\"/\")[-1]\n", + "\n", + "print(dataset_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_datasets_import:migration,new" + }, + "source": [ + "### [projects.locations.datasets.import](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.datasets/import)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "FrIGGdbC1sik" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "L4UZJnOM1sik" + }, + "outputs": [], + "source": [ + "LABEL_SCHEMA = IMPORT_SCHEMA_TEXT_SENTIMENT\n", + "\n", + "import_config = {\n", + " \"gcs_source\": {\"uris\": [IMPORT_FILE]},\n", + " \"import_schema_uri\": LABEL_SCHEMA,\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.ImportDataRequest(name=dataset_id, import_configs=[import_config]).__dict__[\n", + " \"_pb\"\n", + " ]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/datasets/3047617533776494592\",\n", + " \"importConfigs\": [\n", + " {\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://cloud-samples-data/language/claritin-split.csv\"\n", + " ]\n", + " },\n", + " \"importSchemaUri\": \"gs://google-cloud-aiplatform/schema/dataset/ioformat/text_sentiment_io_format_1.0.0.yaml\"\n", + " }\n", + " ]\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "klKZuShs1sik" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "KURUWPTj1sil" + }, + "outputs": [], + "source": [ + "request = clients[\"dataset\"].import_data(\n", + " name=dataset_id, import_configs=[import_config]\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Ggs4F4QJ1sil" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "4amvwklf1sil" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_create_and_deploy_model:migration" + }, + "source": [ + "## Train a model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0oqIBOSnJjkW" + }, + "source": [ + "### [projects.locations.trainingPipelines.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.trainingPipelines/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "rqTHZlwM1sil" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Y2EnBqsW1sim" + }, + "outputs": [], + "source": [ + "TRAINING_SCHEMA = TRAINING_TEXT_SENTIMENT_SCHEMA\n", + "\n", + "task = Value(struct_value=Struct(fields={\"sentiment_max\": Value(number_value=4)}))\n", + "\n", + "training_pipeline = {\n", + " \"display_name\": \"claritin_\" + TIMESTAMP,\n", + " \"input_data_config\": {\"dataset_id\": dataset_short_id},\n", + " \"model_to_upload\": {\"display_name\": \"claritin_\" + TIMESTAMP},\n", + " \"training_task_definition\": TRAINING_SCHEMA,\n", + " \"training_task_inputs\": task,\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateTrainingPipelineRequest(\n", + " parent=PARENT,\n", + " training_pipeline=training_pipeline,\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"trainingPipeline\": {\n", + " \"displayName\": \"claritin_20210301212135\",\n", + " \"inputDataConfig\": {\n", + " \"datasetId\": \"3047617533776494592\"\n", + " },\n", + " \"trainingTaskDefinition\": \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_text_sentiment_1.0.0.yaml\",\n", + " \"trainingTaskInputs\": {\n", + " \"sentiment_max\": 4.0\n", + " },\n", + " \"modelToUpload\": {\n", + " \"displayName\": \"claritin_20210301212135\"\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ynb6GJWQ1sim" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "LahVyHQD1sim" + }, + "outputs": [], + "source": [ + "request = clients[\"pipeline\"].create_training_pipeline(\n", + " parent=PARENT, training_pipeline=training_pipeline\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "xES6wtah1sim" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Kw_d_K7j1sim" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/trainingPipelines/5293990158067040256\",\n", + " \"displayName\": \"claritin_20210301212135\",\n", + " \"inputDataConfig\": {\n", + " \"datasetId\": \"3047617533776494592\"\n", + " },\n", + " \"trainingTaskDefinition\": \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_text_sentiment_1.0.0.yaml\",\n", + " \"trainingTaskInputs\": {\n", + " \"sentimentMax\": 4.0\n", + " },\n", + " \"modelToUpload\": {\n", + " \"displayName\": \"claritin_20210301212135\"\n", + " },\n", + " \"state\": \"PIPELINE_STATE_PENDING\",\n", + " \"createTime\": \"2021-03-01T21:32:03.085444Z\",\n", + " \"updateTime\": \"2021-03-01T21:32:03.085444Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "54bab5051996" + }, + "outputs": [], + "source": [ + "# The full unique ID for the training pipeline\n", + "training_pipeline_id = request.name\n", + "# The short numeric ID for the training pipeline\n", + "training_pipeline_short_id = training_pipeline_id.split(\"/\")[-1]\n", + "\n", + "print(training_pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "QJsV1xoSJjkW" + }, + "source": [ + "### [projects.locations.trainingPipelines.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.trainingPipelines/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wZZJzTBo1sin" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "qBKk5fXY1sin" + }, + "outputs": [], + "source": [ + "request = clients[\"pipeline\"].get_training_pipeline(name=training_pipeline_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "RCLZYC_T1sin" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "P8qJbudB1sin" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/trainingPipelines/5293990158067040256\",\n", + " \"displayName\": \"claritin_20210301212135\",\n", + " \"inputDataConfig\": {\n", + " \"datasetId\": \"3047617533776494592\"\n", + " },\n", + " \"trainingTaskDefinition\": \"gs://google-cloud-aiplatform/schema/trainingjob/definition/automl_text_sentiment_1.0.0.yaml\",\n", + " \"trainingTaskInputs\": {\n", + " \"sentimentMax\": 4.0\n", + " },\n", + " \"modelToUpload\": {\n", + " \"displayName\": \"claritin_20210301212135\"\n", + " },\n", + " \"state\": \"PIPELINE_STATE_PENDING\",\n", + " \"createTime\": \"2021-03-01T21:32:03.085444Z\",\n", + " \"updateTime\": \"2021-03-01T21:32:03.085444Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = clients[\"pipeline\"].get_training_pipeline(name=training_pipeline_id)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " break\n", + " else:\n", + " model_id = response.model_to_upload.name\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(20)\n", + "\n", + "print(model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_evaluate_the_model:migration" + }, + "source": [ + "## Evaluate the model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "BJNzhELXJjkY" + }, + "source": [ + "### [projects.locations.models.evaluations.list](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.models.evaluations/list)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "HfPQn-dg1sio" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cK7Vpv9i1sio" + }, + "outputs": [], + "source": [ + "request = clients[\"model\"].list_model_evaluations(parent=model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "OQDH5S_d1sio" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "prEvwExMmDy8" + }, + "outputs": [], + "source": [ + "model_evaluations = [json.loads(MessageToJson(mel.__dict__[\"_pb\"])) for mel in request]\n", + "\n", + "# The evaluation slice\n", + "evaluation_slice = request.model_evaluations[0].name\n", + "\n", + "print(json.dumps(model_evaluations, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "[\n", + " {\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/5497364624833511424/evaluations/412684097299677184\",\n", + " \"metricsSchemaUri\": \"gs://google-cloud-aiplatform/schema/modelevaluation/text_sentiment_metrics_1.0.0.yaml\",\n", + " \"metrics\": {\n", + " \"linearKappa\": 0.4001057,\n", + " \"quadraticKappa\": 0.48378703,\n", + " \"precision\": 0.59030837,\n", + " \"confusionMatrix\": {\n", + " \"rows\": [\n", + " [\n", + " 3.0,\n", + " 4.0,\n", + " 1.0,\n", + " 1.0,\n", + " 0.0\n", + " ],\n", + " [\n", + " 2.0,\n", + " 19.0,\n", + " 25.0,\n", + " 18.0,\n", + " 0.0\n", + " ],\n", + " [\n", + " 0.0,\n", + " 10.0,\n", + " 72.0,\n", + " 57.0,\n", + " 0.0\n", + " ],\n", + " [\n", + " 0.0,\n", + " 16.0,\n", + " 34.0,\n", + " 169.0,\n", + " 5.0\n", + " ],\n", + " [\n", + " 0.0,\n", + " 0.0,\n", + " 2.0,\n", + " 11.0,\n", + " 5.0\n", + " ]\n", + " ],\n", + " \"annotationSpecs\": [\n", + " {\n", + " \"displayName\": \"0\",\n", + " \"id\": \"7302033741432487936\"\n", + " },\n", + " {\n", + " \"id\": \"1537426218398253056\",\n", + " \"displayName\": \"1\"\n", + " },\n", + " {\n", + " \"displayName\": \"2\",\n", + " \"id\": \"6149112236825640960\"\n", + " },\n", + " {\n", + " \"displayName\": \"3\",\n", + " \"id\": \"3843269227611947008\"\n", + " },\n", + " {\n", + " \"displayName\": \"4\",\n", + " \"id\": \"8454955246039334912\"\n", + " }\n", + " ]\n", + " },\n", + " \"meanAbsoluteError\": 0.4955947,\n", + " \"f1Score\": 0.59030837,\n", + " \"recall\": 0.59030837,\n", + " \"meanSquaredError\": 0.67180616\n", + " },\n", + " \"createTime\": \"2021-03-02T01:41:13.130713Z\",\n", + " \"sliceDimensions\": [\n", + " \"annotationSpec\"\n", + " ]\n", + " }\n", + "]\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wr2JuU5nJjka" + }, + "source": [ + "### [projects.locations.models.evaluations.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.models.evaluations/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "kh15ZsBj1sip" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "eRbV7hzH1sip" + }, + "outputs": [], + "source": [ + "request = clients[\"model\"].get_model_evaluation(name=evaluation_slice)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "KQLWoPmU1sip" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "fo2fAEf61sip" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPupiwqN_jAB" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/models/5497364624833511424/evaluations/412684097299677184\",\n", + " \"metricsSchemaUri\": \"gs://google-cloud-aiplatform/schema/modelevaluation/text_sentiment_metrics_1.0.0.yaml\",\n", + " \"metrics\": {\n", + " \"meanSquaredError\": 0.67180616,\n", + " \"linearKappa\": 0.4001057,\n", + " \"precision\": 0.59030837,\n", + " \"recall\": 0.59030837,\n", + " \"confusionMatrix\": {\n", + " \"annotationSpecs\": [\n", + " {\n", + " \"id\": \"7302033741432487936\",\n", + " \"displayName\": \"0\"\n", + " },\n", + " {\n", + " \"id\": \"1537426218398253056\",\n", + " \"displayName\": \"1\"\n", + " },\n", + " {\n", + " \"displayName\": \"2\",\n", + " \"id\": \"6149112236825640960\"\n", + " },\n", + " {\n", + " \"id\": \"3843269227611947008\",\n", + " \"displayName\": \"3\"\n", + " },\n", + " {\n", + " \"id\": \"8454955246039334912\",\n", + " \"displayName\": \"4\"\n", + " }\n", + " ],\n", + " \"rows\": [\n", + " [\n", + " 3.0,\n", + " 4.0,\n", + " 1.0,\n", + " 1.0,\n", + " 0.0\n", + " ],\n", + " [\n", + " 2.0,\n", + " 19.0,\n", + " 25.0,\n", + " 18.0,\n", + " 0.0\n", + " ],\n", + " [\n", + " 0.0,\n", + " 10.0,\n", + " 72.0,\n", + " 57.0,\n", + " 0.0\n", + " ],\n", + " [\n", + " 0.0,\n", + " 16.0,\n", + " 34.0,\n", + " 169.0,\n", + " 5.0\n", + " ],\n", + " [\n", + " 0.0,\n", + " 0.0,\n", + " 2.0,\n", + " 11.0,\n", + " 5.0\n", + " ]\n", + " ]\n", + " },\n", + " \"meanAbsoluteError\": 0.4955947,\n", + " \"quadraticKappa\": 0.48378703,\n", + " \"f1Score\": 0.59030837\n", + " },\n", + " \"createTime\": \"2021-03-02T01:41:13.130713Z\",\n", + " \"sliceDimensions\": [\n", + " \"annotationSpec\"\n", + " ]\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "8T8qLCuR1sip" + }, + "source": [ + "## Make batch predictions" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "c660ee11e92e" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "Let's now make a batch input file, which you will store in your local Cloud Storage bucket. The batch input file can be either CSV or JSONL. You will use JSONL in this tutorial. For JSONL file, you make one dictionary entry per line for each data item (instance). The dictionary contains the key/value pairs:\n", + "\n", + "- `content`: The Cloud Storage path to the image.\n", + "- `mime_type`: The content type. In our example, it is an `test/plain` file." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bPax2DdN1siq" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "import tensorflow as tf\n", + "\n", + "test_data = ! gsutil cat $IMPORT_FILE | head -n1\n", + "\n", + "test_item = str(test_data[0]).split(\",\")[1]\n", + "test_label = str(test_data[0]).split(\",\")[2]\n", + "\n", + "gcs_test_item = \"gs://\" + BUCKET_NAME + \"/test.txt\"\n", + "with tf.io.gfile.GFile(gcs_test_item, \"w\") as f:\n", + " f.write(test_item + \"\\n\")\n", + "\n", + "gcs_input_uri = \"gs://\" + BUCKET_NAME + \"/test.jsonl\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " data = {\"content\": gcs_test_item, \"mime_type\": \"text/plain\"}\n", + " f.write(json.dumps(data) + \"\\n\")\n", + "\n", + "! gsutil cat $gcs_input_uri\n", + "! gsutil cat $gcs_test_item" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\"content\": \"gs://migration-ucaip-trainingaip-20210301212135/test.txt\", \"mime_type\": \"text/plain\"}\n", + "@freewrytin God is way too good for Claritin\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tnqYDtYF1siq" + }, + "source": [ + "### [projects.locations.batchPredictionJobs.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.batchPredictionJobs/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ozaMOwdNJjkT" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "XXYA8Ns21siq" + }, + "outputs": [], + "source": [ + "batch_prediction_job = {\n", + " \"display_name\": \"claritin_\" + TIMESTAMP,\n", + " \"model\": model_id,\n", + " \"input_config\": {\n", + " \"instances_format\": \"jsonl\",\n", + " \"gcs_source\": {\"uris\": [gcs_input_uri]},\n", + " },\n", + " \"output_config\": {\n", + " \"predictions_format\": \"jsonl\",\n", + " \"gcs_destination\": {\n", + " \"output_uri_prefix\": \"gs://\" + f\"{BUCKET_NAME}/batch_output/\"\n", + " },\n", + " },\n", + " \"dedicated_resources\": {\n", + " \"machine_spec\": {\"machine_type\": \"n1-standard-2\", \"accelerator_count\": 0},\n", + " \"starting_replica_count\": 1,\n", + " \"max_replica_count\": 1,\n", + " },\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateBatchPredictionJobRequest(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"batchPredictionJob\": {\n", + " \"displayName\": \"claritin_20210301212135\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/5497364624833511424\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210301212135/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210301212135/batch_output/\"\n", + " }\n", + " },\n", + " \"dedicatedResources\": {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-2\"\n", + " },\n", + " \"startingReplicaCount\": 1,\n", + " \"maxReplicaCount\": 1\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "NF7BDJ9v1siq" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "vNaCWhY91siq" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].create_batch_prediction_job(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "n78sGHxZ1sir" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "VjX3nEZv1sir" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/batchPredictionJobs/3543215802926759936\",\n", + " \"displayName\": \"claritin_20210301212135\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/5497364624833511424\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210301212135/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210301212135/batch_output/\"\n", + " }\n", + " },\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"completionStats\": {\n", + " \"incompleteCount\": \"-1\"\n", + " },\n", + " \"createTime\": \"2021-03-02T01:41:31.919764Z\",\n", + " \"updateTime\": \"2021-03-02T01:41:31.919764Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "7d448b1cc8c1" + }, + "outputs": [], + "source": [ + "# The fully qualified ID for the batch job\n", + "batch_job_id = request.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split(\"/\")[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "mZ8NQGM31sir" + }, + "source": [ + "### [projects.locations.batchPredictionJobs.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.batchPredictionJobs/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "zZOSE1961sir" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "kEMq5lPi1sir" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].get_batch_prediction_job(name=batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "9cgeT_7B1sir" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "OT3wuCUw1sis" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/batchPredictionJobs/3543215802926759936\",\n", + " \"displayName\": \"claritin_20210301212135\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/5497364624833511424\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210301212135/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210301212135/batch_output/\"\n", + " }\n", + " },\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"completionStats\": {\n", + " \"incompleteCount\": \"-1\"\n", + " },\n", + " \"createTime\": \"2021-03-02T01:41:31.919764Z\",\n", + " \"updateTime\": \"2021-03-02T01:41:31.919764Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "3cdf7f0526ec" + }, + "outputs": [], + "source": [ + "def get_latest_predictions(gcs_out_dir):\n", + " \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n", + " folders = !gsutil ls $gcs_out_dir\n", + " latest = \"\"\n", + " for folder in folders:\n", + " subfolder = folder.split(\"/\")[-2]\n", + " if subfolder.startswith(\"prediction-\"):\n", + " if subfolder > latest:\n", + " latest = folder[:-1]\n", + " return latest\n", + "\n", + "\n", + "while True:\n", + " response = clients[\"job\"].get_batch_prediction_job(name=batch_job_id)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", response.state)\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " folder = get_latest_predictions(\n", + " response.output_config.gcs_destination.output_uri_prefix\n", + " )\n", + " ! gsutil ls $folder/prediction*.jsonl\n", + "\n", + " ! gsutil cat $folder/prediction*.jsonl\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "gs://migration-ucaip-trainingaip-20210301212135/batch_output/prediction-claritin_20210301212135-2021-03-02T01:41:31.705301Z/predictions_00001.jsonl\n", + "{\"instance\":{\"content\":\"gs://migration-ucaip-trainingaip-20210301212135/test.txt\",\"mimeType\":\"text/plain\"},\"prediction\":{\"sentiment\":2}}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aeAz5yPa1sis" + }, + "source": [ + "## Make online predictions" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "h4_aWo7-1sis" + }, + "source": [ + "### Prepare data item for online prediction" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "86h1FUtA1sis" + }, + "outputs": [], + "source": [ + "test_data = ! gsutil cat $IMPORT_FILE | head -n1\n", + "\n", + "test_item = str(test_data[0]).split(\",\")[1]\n", + "test_label = str(test_data[0]).split(\",\")[2]\n", + "\n", + "print((test_item, test_label))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "COwVZtxhJjkW" + }, + "source": [ + "### [projects.locations.endpoints.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "NI50tf4Q1sit" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tdNxDl9U1sit" + }, + "outputs": [], + "source": [ + "endpoint = {\"display_name\": \"claritin_\" + TIMESTAMP}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateEndpointRequest(parent=PARENT, endpoint=endpoint).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"endpoint\": {\n", + " \"displayName\": \"claritin_20210301212135\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "o-3TRQXY1sit" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tha965qB1sit" + }, + "outputs": [], + "source": [ + "request = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "zNmlwwWh1sit" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "AWpQw1G61sit" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/endpoints/45845236831748096\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "c469d14a26ea" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gpnVJS3AJjkW" + }, + "source": [ + "### [projects.locations.endpoints.deployModel](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/deployModel)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "fRWgao0Q1siu" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "RV9Ytz4o1siu" + }, + "outputs": [], + "source": [ + "deployed_model = {\n", + " \"model\": model_id,\n", + " \"display_name\": \"claritin_\" + TIMESTAMP,\n", + " \"automatic_resources\": {\"min_replica_count\": 1, \"max_replica_count\": 1},\n", + "}\n", + "\n", + "traffic_split = {\"0\": 100}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.DeployModelRequest(\n", + " endpoint=endpoint_id,\n", + " deployed_model=deployed_model,\n", + " traffic_split=traffic_split,\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"endpoint\": \"projects/116273516712/locations/us-central1/endpoints/45845236831748096\",\n", + " \"deployedModel\": {\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/5497364624833511424\",\n", + " \"displayName\": \"claritin_20210301212135\",\n", + " \"automaticResources\": {\n", + " \"minReplicaCount\": 1,\n", + " \"maxReplicaCount\": 1\n", + " }\n", + " },\n", + " \"trafficSplit\": {\n", + " \"0\": 100\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "J7bV5doB1siu" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "28zuTgn31siu" + }, + "outputs": [], + "source": [ + "request = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint_id, deployed_model=deployed_model, traffic_split=traffic_split\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Gi-yN6VJ1siv" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "_CWhWunf1siv" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"deployedModel\": {\n", + " \"id\": \"6669232913810194432\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "629d19cbcf39" + }, + "outputs": [], + "source": [ + "# The unique ID for the deployed model\n", + "deployed_model_id = result.deployed_model.id\n", + "\n", + "print(deployed_model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "6a0-OOQ61siv" + }, + "source": [ + "### [projects.locations.endpoints.predict](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/predict)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ABoH16AE1siw" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "qFzxI6uY1siw" + }, + "outputs": [], + "source": [ + "instances_list = [{\"content\": test_item}]\n", + "instances = [json_format.ParseDict(s, Value()) for s in instances_list]\n", + "\n", + "request = aip.PredictRequest(endpoint=endpoint_id)\n", + "request.instances.append(instances)\n", + "\n", + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"endpoint\": \"projects/116273516712/locations/us-central1/endpoints/45845236831748096\",\n", + " \"instances\": [\n", + " [\n", + " {\n", + " \"content\": \"@freewrytin God is way too good for Claritin\"\n", + " }\n", + " ]\n", + " ]\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "5HyneZhf1siw" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "mkC_HbE11siw" + }, + "outputs": [], + "source": [ + "request = clients[\"prediction\"].predict(endpoint=endpoint_id, instances=instances)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "9KIx8hQZ1siw" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Tq2gITNu1six" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"predictions\": [\n", + " {\n", + " \"sentiment\": 2.0\n", + " }\n", + " ],\n", + " \"deployedModelId\": \"6669232913810194432\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new" + }, + "source": [ + "### [projects.locations.endpoints.undeployModel](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/undeployModel)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "KrVZz6Uw_jAp" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_undeploymodel:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint_id, deployed_model_id=deployed_model_id, traffic_split={}\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wvFK-kir_jAq" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "GvajSw-Y_jAq" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bQ-VVaSxJjkd" + }, + "source": [ + "# Cleaning up\r\n", + "\r\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\r\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\r\n", + "\r\n", + "Otherwise, you can delete the individual resources you created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ZLFTpeDL1six" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_pipeline = True\n", + "delete_batchjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the dataset using the Vertex AI fully qualified identifier for the dataset\n", + "try:\n", + " if delete_dataset:\n", + " clients[\"dataset\"].delete_dataset(name=dataset_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex AI fully qualified identifier for the model\n", + "try:\n", + " if delete_model:\n", + " clients[\"model\"].delete_model(name=model_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex AI fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint:\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the training pipeline using the Vertex AI fully qualified identifier for the training pipeline\n", + "try:\n", + " if delete_pipeline:\n", + " clients[\"pipeline\"].delete_training_pipeline(name=training_pipeline_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex AI fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob:\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r gs://$BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "UJ8 unified AutoML Natural Language - Text Sentiment Analysis.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/migration/UJ9 legacy Custom Training Prebuilt Container XGBoost.ipynb b/notebooks/community/migration/UJ9 legacy Custom Training Prebuilt Container XGBoost.ipynb new file mode 100644 index 000000000..fdb36de1b --- /dev/null +++ b/notebooks/community/migration/UJ9 legacy Custom Training Prebuilt Container XGBoost.ipynb @@ -0,0 +1,1542 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title:migration,new" + }, + "source": [ + "# Vertex SDK: Train and deploy an XGBoost model with pre-built containers (formerly hosted runtimes)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the Google *cloud-storage* library as well.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "KyQ2Ul4WaTRU" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-storage" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the Kernel\n", + "\n", + "Once you've installed the Vertex SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "mkVf4egRaTRW" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "if not os.getenv(\"AUTORUN\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU run-time\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your GCP project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a GCP project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebooks.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend when possible, to choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You cannot use a Multi-Regional Storage bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see [Region support for Vertex services]()\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ZOKWWPEgaTRZ" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "s5-9blJNaTRa" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your GCP account\n", + "\n", + "**If you are using Google Cloud Notebooks**, your environment is already\n", + "authenticated. Skip this step.\n", + "\n", + "*Note: If you are on an Vertex notebook and run the cell, the cell knows to skip executing the authentication steps.*\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "MkU__HblaTRa" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your Google Cloud account. This provides access\n", + "# to your Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Vertex, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this tutorial in a notebook locally, replace the string\n", + " # below with the path to your service account key and run this cell to\n", + " # authenticate your Google Cloud account.\n", + " else:\n", + " %env GOOGLE_APPLICATION_CREDENTIALS your_path_to_credentials.json\n", + "\n", + " # Log in to your account on Google Cloud\n", + " ! gcloud auth login" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n", + " BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "zBAe4hnoaTRc" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "1sFpszMlaTRd" + }, + "outputs": [], + "source": [ + "! gsutil ls -al gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip" + }, + "source": [ + "#### Import Vertex SDK\n", + "\n", + "Import the Vertex SDK into our Python environment.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ABxRhG9HaTRe" + }, + "outputs": [], + "source": [ + "import json\n", + "import time\n", + "\n", + "from googleapiclient import discovery" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex constants\n", + "\n", + "Setup up the following constants for Vertex:\n", + "\n", + "- `PARENT`: The Vertex location root path for dataset, model and endpoint resources.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "PiqsVnv0aTRe" + }, + "outputs": [], + "source": [ + "# Vertex location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients" + }, + "source": [ + "## Clients\n", + "\n", + "The Vertex SDK works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the server (Vertex).\n", + "\n", + "You will use several clients in this tutorial, so set them all up upfront.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "f35VsXENaTRf" + }, + "outputs": [], + "source": [ + "client = discovery.build(\"ml\", \"v1\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Y94mDnSvaTRf" + }, + "source": [ + "## Prepare a trainer script" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "nC85LEiAaTRf" + }, + "source": [ + "### Package assembly" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "jUqjiBWKaTRg" + }, + "outputs": [], + "source": [ + "# Make folder for python training script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\\n", + "tag_build =\\n\\\n", + "tag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\\n", + "setuptools.setup(\\n\\\n", + " install_requires=[\\n\\\n", + " ],\\n\\\n", + " packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\\n", + "Name: Custom XGBoost Iris\\n\\\n", + "Version: 0.0.0\\n\\\n", + "Summary: Demonstration training script\\n\\\n", + "Home-page: www.google.com\\n\\\n", + "Author: Google\\n\\\n", + "Author-email: aferlitsch@google.com\\n\\\n", + "License: Public\\n\\\n", + "Description: Demo\\n\\\n", + "Platform: Vertex AI\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "swLUnjkOaTRg" + }, + "source": [ + "### Task.py contents" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "49ryRrhuaTRg" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Single Instance Training for Iris\n", + "\n", + "import datetime\n", + "import os\n", + "import subprocess\n", + "import sys\n", + "import pandas as pd\n", + "import xgboost as xgb\n", + "\n", + "import argparse\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv('AIP_MODEL_DIR'), type=str, help='Model dir.')\n", + "args = parser.parse_args()\n", + "\n", + "# Download data\n", + "iris_data_filename = 'iris_data.csv'\n", + "iris_target_filename = 'iris_target.csv'\n", + "data_dir = 'gs://cloud-samples-data/ai-platform/iris'\n", + "\n", + "# gsutil outputs everything to stderr so we need to divert it to stdout.\n", + "subprocess.check_call(['gsutil', 'cp', os.path.join(data_dir,\n", + " iris_data_filename),\n", + " iris_data_filename], stderr=sys.stdout)\n", + "subprocess.check_call(['gsutil', 'cp', os.path.join(data_dir,\n", + " iris_target_filename),\n", + " iris_target_filename], stderr=sys.stdout)\n", + "\n", + "\n", + "# Load data into pandas, then use `.values` to get NumPy arrays\n", + "iris_data = pd.read_csv(iris_data_filename).values\n", + "iris_target = pd.read_csv(iris_target_filename).values\n", + "\n", + "# Convert one-column 2D array into 1D array for use with XGBoost\n", + "iris_target = iris_target.reshape((iris_target.size,))\n", + "\n", + "\n", + "# Load data into DMatrix object\n", + "dtrain = xgb.DMatrix(iris_data, label=iris_target)\n", + "\n", + "# Train XGBoost model\n", + "bst = xgb.train({}, dtrain, 20)\n", + "\n", + "# Export the classifier to a file\n", + "model_filename = 'model.bst'\n", + "bst.save_model(model_filename)\n", + "\n", + "\n", + "# Upload the saved model file to Cloud Storage\n", + "gcs_model_path = os.path.join(args.model_dir, model_filename)\n", + "subprocess.check_call(['gsutil', 'cp', model_filename, gcs_model_path],\n", + " stderr=sys.stdout)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ZK8L6hTNaTRg" + }, + "source": [ + "### Store training script on your Cloud Storage bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "I47eBUkzaTRh" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz gs://$BUCKET_NAME/iris.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_create_and_deploy_model:migration" + }, + "source": [ + "## Train a model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0oqIBOSnJjkW" + }, + "source": [ + "### [projects.jobs.create](https://cloud.google.com/ai-platform/training/docs/reference/rest/v1/projects.jobs/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "v7mIfIzWaTRi" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "8_84f21UaTRi" + }, + "outputs": [], + "source": [ + "JOB_NAME = \"custom_job_XGB\" + TIMESTAMP\n", + "\n", + "training_input = {\n", + " \"scaleTier\": \"BASIC\",\n", + " \"packageUris\": [\"gs://\" + BUCKET_NAME + \"/iris.tar.gz\"],\n", + " \"pythonModule\": \"trainer.task\",\n", + " \"args\": [\"--model-dir=\" + \"gs://{}/{}\".format(BUCKET_NAME, JOB_NAME)],\n", + " \"region\": REGION,\n", + " \"runtimeVersion\": \"2.4\",\n", + " \"pythonVersion\": \"3.7\",\n", + "}\n", + "\n", + "body = {\"jobId\": JOB_NAME, \"trainingInput\": training_input}\n", + "\n", + "request = client.projects().jobs().create(parent=\"projects/\" + PROJECT_ID)\n", + "request.body = body\n", + "\n", + "print(json.dumps(json.loads(request.to_json()), indent=2))\n", + "\n", + "request = client.projects().jobs().create(parent=\"projects/\" + PROJECT_ID, body=body)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"uri\": \"https://ml.googleapis.com/v1/projects/migration-ucaip-training/jobs?alt=json\",\n", + " \"method\": \"POST\",\n", + " \"body\": {\n", + " \"jobId\": \"custom_job_XGB20210302154841\",\n", + " \"trainingInput\": {\n", + " \"scaleTier\": \"BASIC\",\n", + " \"packageUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210302154841/iris.tar.gz\"\n", + " ],\n", + " \"pythonModule\": \"trainer.task\",\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210302154841/custom_job_XGB20210302154841\"\n", + " ],\n", + " \"region\": \"us-central1\",\n", + " \"runtimeVersion\": \"2.4\",\n", + " \"pythonVersion\": \"3.7\"\n", + " }\n", + " },\n", + " \"headers\": {\n", + " \"accept\": \"application/json\",\n", + " \"accept-encoding\": \"gzip, deflate\",\n", + " \"user-agent\": \"(gzip)\",\n", + " \"x-goog-api-client\": \"gdcl/1.12.8 gl-python/3.7.8\"\n", + " },\n", + " \"methodId\": \"ml.projects.jobs.create\",\n", + " \"resumable\": null,\n", + " \"response_callbacks\": [],\n", + " \"_in_error_state\": false,\n", + " \"body_size\": 0,\n", + " \"resumable_uri\": null,\n", + " \"resumable_progress\": 0\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tZySnxxuaTRi" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "_H3xV191aTRj" + }, + "outputs": [], + "source": [ + "result = request.execute()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "vbwW_N-eaTRj" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Z6D71QmqaTRj" + }, + "outputs": [], + "source": [ + "print(json.dumps(result, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "5xOLmTpKaTRj" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"jobId\": \"custom_job_XGB20210302154841\",\n", + " \"trainingInput\": {\n", + " \"packageUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210302154841/iris.tar.gz\"\n", + " ],\n", + " \"pythonModule\": \"trainer.task\",\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210302154841/custom_job_XGB20210302154841\"\n", + " ],\n", + " \"region\": \"us-central1\",\n", + " \"runtimeVersion\": \"2.4\",\n", + " \"pythonVersion\": \"3.7\"\n", + " },\n", + " \"createTime\": \"2021-03-02T15:50:12Z\",\n", + " \"state\": \"QUEUED\",\n", + " \"trainingOutput\": {},\n", + " \"etag\": \"PmcK2JEDnJM=\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "training_pipeline_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The short numeric ID for the custom training job\n", + "custom_training_short_id = result[\"jobId\"]\n", + "# The full unique ID for the custom training job\n", + "custom_training_id = \"projects/\" + PROJECT_ID + \"/jobs/\" + result[\"jobId\"]\n", + "\n", + "print(custom_training_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "EXSBAbs_aTRk" + }, + "source": [ + "### [projects.jobs.get](https://cloud.google.com/ai-platform/training/docs/reference/rest/v1/projects.jobs/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "qQawJaAEaTRk" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "a_9aXJKHaTRk" + }, + "outputs": [], + "source": [ + "request = client.projects().jobs().get(name=custom_training_id)\n", + "\n", + "result = request.execute()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bCd7K6WjaTRk" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "xI_osgwsaTRl" + }, + "outputs": [], + "source": [ + "print(json.dumps(result, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "EkwQk94iaTRl" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"jobId\": \"custom_job_XGB20210302154841\",\n", + " \"trainingInput\": {\n", + " \"packageUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210302154841/iris.tar.gz\"\n", + " ],\n", + " \"pythonModule\": \"trainer.task\",\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210302154841/custom_job_XGB20210302154841\"\n", + " ],\n", + " \"region\": \"us-central1\",\n", + " \"runtimeVersion\": \"2.4\",\n", + " \"pythonVersion\": \"3.7\"\n", + " },\n", + " \"createTime\": \"2021-03-02T15:50:12Z\",\n", + " \"state\": \"PREPARING\",\n", + " \"trainingOutput\": {},\n", + " \"etag\": \"L+085Kgm1Wo=\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = client.projects().jobs().get(name=custom_training_id).execute()\n", + "\n", + " if response[\"state\"] != \"SUCCEEDED\":\n", + " print(\"Training job has not completed:\", response[\"state\"])\n", + " if response[\"state\"] == \"FAILED\":\n", + " break\n", + " else:\n", + " break\n", + " time.sleep(60)\n", + "\n", + "# model artifact output directory on Google Cloud Storage\n", + "model_artifact_dir = response[\"trainingInput\"][\"args\"][0].split(\"=\")[-1]\n", + "print(\"artifact location \" + model_artifact_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "_2zUfQ3HaTRl" + }, + "source": [ + "## Deploy the model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "uuMhstTYaTRl" + }, + "source": [ + "### [projects.models.create](https://cloud.google.com/ai-platform/prediction/docs/reference/rest/v1/projects.models/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "UGvtc-O6aTRm" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pWfw2rPGaTRm" + }, + "outputs": [], + "source": [ + "body = {\"name\": \"custom_job_XGB\" + TIMESTAMP}\n", + "\n", + "request = client.projects().models().create(parent=\"projects/\" + PROJECT_ID)\n", + "request.body = json.loads(json.dumps(body, indent=2))\n", + "\n", + "print(json.dumps(json.loads(request.to_json()), indent=2))\n", + "\n", + "request = client.projects().models().create(parent=\"projects/\" + PROJECT_ID, body=body)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "zGiLpVSKaTRm" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"uri\": \"https://ml.googleapis.com/v1/projects/migration-ucaip-training/models?alt=json\",\n", + " \"method\": \"POST\",\n", + " \"body\": {\n", + " \"name\": \"custom_job_XGB20210302154841\"\n", + " },\n", + " \"headers\": {\n", + " \"accept\": \"application/json\",\n", + " \"accept-encoding\": \"gzip, deflate\",\n", + " \"user-agent\": \"(gzip)\",\n", + " \"x-goog-api-client\": \"gdcl/1.12.8 gl-python/3.7.8\"\n", + " },\n", + " \"methodId\": \"ml.projects.models.create\",\n", + " \"resumable\": null,\n", + " \"response_callbacks\": [],\n", + " \"_in_error_state\": false,\n", + " \"body_size\": 0,\n", + " \"resumable_uri\": null,\n", + " \"resumable_progress\": 0\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "VDkiEXHsaTRm" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dQhpRTy6aTRm" + }, + "outputs": [], + "source": [ + "result = request.execute()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "L6l6Li59aTRn" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "NcOYGasraTRn" + }, + "outputs": [], + "source": [ + "print(json.dumps(result, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "8Au9SgRdaTRn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/migration-ucaip-training/models/custom_job_XGB20210302154841\",\n", + " \"regions\": [\n", + " \"us-central1\"\n", + " ],\n", + " \"etag\": \"4gQZjQgH2sc=\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "30f7R83UaTRn" + }, + "outputs": [], + "source": [ + "model_id = result[\"name\"]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tXR5mKGfaTRo" + }, + "source": [ + "### [projects.models.versions.create](https://cloud.google.com/ai-platform/prediction/docs/reference/rest/v1/projects.models.versions/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ZJhWlWJ0aTRo" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "973d-m48aTRo" + }, + "outputs": [], + "source": [ + "version = {\n", + " \"name\": \"custom_job_XGB\" + TIMESTAMP,\n", + " \"deploymentUri\": model_artifact_dir,\n", + " \"runtimeVersion\": \"2.1\",\n", + " \"framework\": \"XGBOOST\",\n", + " \"pythonVersion\": \"3.7\",\n", + " \"machineType\": \"mls1-c1-m2\",\n", + "}\n", + "\n", + "request = (\n", + " client.projects()\n", + " .models()\n", + " .versions()\n", + " .create(\n", + " parent=model_id,\n", + " )\n", + ")\n", + "request.body = version\n", + "\n", + "print(json.dumps(json.loads(request.to_json()), indent=2))\n", + "\n", + "request = client.projects().models().versions().create(parent=model_id, body=version)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "1-M1ydRDaTRo" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"uri\": \"https://ml.googleapis.com/v1/projects/migration-ucaip-training/models/custom_job_XGB20210302154841/versions?alt=json\",\n", + " \"method\": \"POST\",\n", + " \"body\": {\n", + " \"name\": \"custom_job_XGB20210302154841\",\n", + " \"deploymentUri\": \"gs://migration-ucaip-trainingaip-20210302154841/custom_job_XGB20210302154841\",\n", + " \"runtimeVersion\": \"2.1\",\n", + " \"framework\": \"XGBOOST\",\n", + " \"pythonVersion\": \"3.7\",\n", + " \"machineType\": \"mls1-c1-m2\"\n", + " },\n", + " \"headers\": {\n", + " \"accept\": \"application/json\",\n", + " \"accept-encoding\": \"gzip, deflate\",\n", + " \"user-agent\": \"(gzip)\",\n", + " \"x-goog-api-client\": \"gdcl/1.12.8 gl-python/3.7.8\"\n", + " },\n", + " \"methodId\": \"ml.projects.models.versions.create\",\n", + " \"resumable\": null,\n", + " \"response_callbacks\": [],\n", + " \"_in_error_state\": false,\n", + " \"body_size\": 0,\n", + " \"resumable_uri\": null,\n", + " \"resumable_progress\": 0\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "9hljh6e1aTRo" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "OLHF4f11aTRo" + }, + "outputs": [], + "source": [ + "result = request.execute()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "8H1G9OfKaTRp" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "9hAOwaLIaTRp" + }, + "outputs": [], + "source": [ + "print(json.dumps(result, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "NM0ypd6JaTRp" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/migration-ucaip-training/operations/create_custom_job_XGB20210302154841_custom_job_XGB20210302154841-1614701495149\",\n", + " \"metadata\": {\n", + " \"@type\": \"type.googleapis.com/google.cloud.ml.v1.OperationMetadata\",\n", + " \"createTime\": \"2021-03-02T16:11:35Z\",\n", + " \"operationType\": \"CREATE_VERSION\",\n", + " \"modelName\": \"projects/migration-ucaip-training/models/custom_job_XGB20210302154841\",\n", + " \"version\": {\n", + " \"name\": \"projects/migration-ucaip-training/models/custom_job_XGB20210302154841/versions/custom_job_XGB20210302154841\",\n", + " \"deploymentUri\": \"gs://migration-ucaip-trainingaip-20210302154841/custom_job_XGB20210302154841\",\n", + " \"createTime\": \"2021-03-02T16:11:35Z\",\n", + " \"runtimeVersion\": \"2.1\",\n", + " \"etag\": \"t71tF0fa60o=\",\n", + " \"framework\": \"XGBOOST\",\n", + " \"machineType\": \"mls1-c1-m2\",\n", + " \"pythonVersion\": \"3.7\"\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "et42NCWeaTRp" + }, + "outputs": [], + "source": [ + "# The full unique ID for the model version\n", + "model_version_name = result[\"metadata\"][\"version\"][\"name\"]\n", + "\n", + "print(model_version_name)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "oW51fpAuaTRq" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = (\n", + " client.projects().models().versions().get(name=model_version_name).execute()\n", + " )\n", + " if response[\"state\"] == \"READY\":\n", + " print(\"Model version created.\")\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_predictions:migration" + }, + "source": [ + "## Make batch predictions\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sIZwfjB0aTRq" + }, + "source": [ + "Batch prediction only supports Tensorflow. FRAMEWORK_SCIKIT_LEARN is not currently available." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "8GHHce4oaTRq" + }, + "source": [ + "## Make online predictions" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_items:automl,icn,csv" + }, + "outputs": [], + "source": [ + "INSTANCES = [[1.4, 1.3, 5.1, 2.8], [1.5, 1.2, 4.7, 2.4]]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "o3nuEi6aaTRr" + }, + "source": [ + "### Prepare file for online prediction" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "anWJoMHbaTRr" + }, + "source": [ + "### [projects.predict](https://cloud.google.com/ai-platform/prediction/docs/reference/rest/v1/projects/predict)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "MsCrcvGxaTRr" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "fJbAj2AOaTRr" + }, + "outputs": [], + "source": [ + "request = client.projects().predict(\n", + " name=model_version_name,\n", + ")\n", + "request.body = json.loads(json.dumps({\"instances\": INSTANCES}, indent=2))\n", + "\n", + "print(json.dumps(json.loads(request.to_json()), indent=2))\n", + "\n", + "request = client.projects().predict(\n", + " name=model_version_name, body={\"instances\": INSTANCES}\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "KbyjXjwaaTRr" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"uri\": \"https://ml.googleapis.com/v1/projects/migration-ucaip-training/models/custom_job_XGB20210302154841/versions/custom_job_XGB20210302154841:predict?alt=json\",\n", + " \"method\": \"POST\",\n", + " \"body\": {\n", + " \"instances\": [\n", + " [\n", + " 1.4,\n", + " 1.3,\n", + " 5.1,\n", + " 2.8\n", + " ],\n", + " [\n", + " 1.5,\n", + " 1.2,\n", + " 4.7,\n", + " 2.4\n", + " ]\n", + " ]\n", + " },\n", + " \"headers\": {\n", + " \"accept\": \"application/json\",\n", + " \"accept-encoding\": \"gzip, deflate\",\n", + " \"user-agent\": \"(gzip)\",\n", + " \"x-goog-api-client\": \"gdcl/1.12.8 gl-python/3.7.8\"\n", + " },\n", + " \"methodId\": \"ml.projects.predict\",\n", + " \"resumable\": null,\n", + " \"response_callbacks\": [],\n", + " \"_in_error_state\": false,\n", + " \"body_size\": 0,\n", + " \"resumable_uri\": null,\n", + " \"resumable_progress\": 0\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "CnrfMAOHaTRs" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Pok1QLZfaTRs" + }, + "outputs": [], + "source": [ + "result = request.execute()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "s9VyPDINaTRs" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "5KxN4GC1aTRs" + }, + "outputs": [], + "source": [ + "print(json.dumps(result, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "iJFHIcmzaTRs" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"predictions\": [\n", + " [],\n", + " []\n", + " ]\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wnJ8nQVZaTRs" + }, + "source": [ + "### [projects.models.versions.delete](https://cloud.google.com/ai-platform/prediction/docs/reference/rest/v1/projects.models.versions/delete)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Kf1HH1GMaTRt" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "alPNxY6NaTRt" + }, + "outputs": [], + "source": [ + "request = client.projects().models().versions().delete(name=model_version_name)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cYpvzoEoaTRt" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ueUWt7BOaTRt" + }, + "outputs": [], + "source": [ + "response = request.execute()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "response:migration" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,request" + }, + "outputs": [], + "source": [ + "print(json.dumps(response, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/migration-ucaip-training/operations/delete_custom_job_XGB20210302154841_custom_job_XGB20210302154841-1614709380234\",\n", + " \"metadata\": {\n", + " \"@type\": \"type.googleapis.com/google.cloud.ml.v1.OperationMetadata\",\n", + " \"createTime\": \"2021-03-02T18:23:00Z\",\n", + " \"operationType\": \"DELETE_VERSION\",\n", + " \"modelName\": \"projects/migration-ucaip-training/models/custom_job_XGB20210302154841\",\n", + " \"version\": {\n", + " \"name\": \"projects/migration-ucaip-training/models/custom_job_XGB20210302154841/versions/custom_job_XGB20210302154841\",\n", + " \"deploymentUri\": \"gs://migration-ucaip-trainingaip-20210302154841/custom_job_XGB20210302154841\",\n", + " \"createTime\": \"2021-03-02T16:11:35Z\",\n", + " \"runtimeVersion\": \"2.1\",\n", + " \"state\": \"READY\",\n", + " \"etag\": \"sBx1RZUe3HQ=\",\n", + " \"framework\": \"XGBOOST\",\n", + " \"machineType\": \"mls1-c1-m2\",\n", + " \"pythonVersion\": \"3.7\"\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup:migration,new" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cfdA_HqraTRu" + }, + "outputs": [], + "source": [ + "delete_model = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the model using the Vertex fully qualified identifier for the model\n", + "try:\n", + " if delete_model:\n", + " client.projects().models().delete(name=model_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r gs://$BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "UJ9 legacy Custom Training Prebuilt Container XGBoost.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/migration/UJ9 unified Custom Training Prebuilt Container XGBoost.ipynb b/notebooks/community/migration/UJ9 unified Custom Training Prebuilt Container XGBoost.ipynb new file mode 100644 index 000000000..0a55375e6 --- /dev/null +++ b/notebooks/community/migration/UJ9 unified Custom Training Prebuilt Container XGBoost.ipynb @@ -0,0 +1,2063 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title:migration,new" + }, + "source": [ + "# Vertex SDK: Train and deploy an XGBoost model with pre-built containers (formerly hosted runtimes)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest (preview) version of Vertex SDK.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-aiplatform --user" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage" + }, + "source": [ + "Install the Google *cloud-storage* library as well.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage" + }, + "outputs": [], + "source": [ + "! pip3 install google-cloud-storage" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart" + }, + "source": [ + "### Restart the Kernel\n", + "\n", + "Once you've installed the Vertex SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "if not os.getenv(\"AUTORUN\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU run-time\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your GCP project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a GCP project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebooks.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex AI. We recommend when possible, to choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You cannot use a Multi-Regional Storage bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see [Region support for Vertex AI services](https://cloud.google.com/vertex-ai/docs/general/locations)\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate" + }, + "source": [ + "### Authenticate your GCP account\n", + "\n", + "**If you are using Google Cloud Notebooks**, your environment is already\n", + "authenticated. Skip this step.\n", + "\n", + "*Note: If you are on an Vertex notebook and run the cell, the cell knows to skip executing the authentication steps.*\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your Google Cloud account. This provides access\n", + "# to your Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Vertex, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this tutorial in a notebook locally, replace the string\n", + " # below with the path to your service account key and run this cell to\n", + " # authenticate your Google Cloud account.\n", + " else:\n", + " %env GOOGLE_APPLICATION_CREDENTIALS your_path_to_credentials.json\n", + "\n", + " # Log in to your account on Google Cloud\n", + " ! gcloud auth login" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:batch_prediction" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "This tutorial is designed to use training data that is in a public Cloud Storage bucket and a local Cloud Storage bucket for your batch predictions. You may alternatively use your own training data that you have stored in a local Cloud Storage bucket.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"[your-bucket-name]\":\n", + " BUCKET_NAME = PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket" + }, + "outputs": [], + "source": [ + "! gsutil ls -al gs://$BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_aip" + }, + "source": [ + "#### Import Vertex SDK\n", + "\n", + "Import the Vertex SDK into our Python environment.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "import time\n", + "\n", + "from google.cloud.aiplatform import gapic as aip\n", + "from google.protobuf.json_format import MessageToJson, ParseDict\n", + "from google.protobuf.struct_pb2 import Struct, Value" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "aip_constants" + }, + "source": [ + "#### Vertex AI constants\n", + "\n", + "Setup up the following constants for Vertex AI:\n", + "\n", + "- `API_ENDPOINT`: The Vertex AI API service endpoint for dataset, model, job, pipeline and endpoint services.\n", + "- `PARENT`: The Vertex AI location root path for dataset, model and endpoint resources.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aip_constants" + }, + "outputs": [], + "source": [ + "# API Endpoint\n", + "API_ENDPOINT = \"{}-aiplatform.googleapis.com\".format(REGION)\n", + "\n", + "# Vertex AI location root path for your dataset, model and endpoint resources\n", + "PARENT = \"projects/\" + PROJECT_ID + \"/locations/\" + REGION" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "clients" + }, + "source": [ + "## Clients\n", + "\n", + "The Vertex SDK works as a client/server model. On your side (the Python script) you will create a client that sends requests and receives responses from the server (Vertex).\n", + "\n", + "You will use several clients in this tutorial, so set them all up upfront.\n", + "\n", + "- Model Service for managed models.\n", + "- Endpoint Service for deployment.\n", + "- Job Service for batch jobs and custom training.\n", + "- Prediction Service for serving. *Note*: Prediction has a different service endpoint.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "clients" + }, + "outputs": [], + "source": [ + "# client options same for all services\n", + "client_options = {\"api_endpoint\": API_ENDPOINT}\n", + "\n", + "\n", + "def create_model_client():\n", + " client = aip.ModelServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_endpoint_client():\n", + " client = aip.EndpointServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_prediction_client():\n", + " client = aip.PredictionServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "def create_job_client():\n", + " client = aip.JobServiceClient(client_options=client_options)\n", + " return client\n", + "\n", + "\n", + "clients = {}\n", + "clients[\"model\"] = create_model_client()\n", + "clients[\"endpoint\"] = create_endpoint_client()\n", + "clients[\"prediction\"] = create_prediction_client()\n", + "clients[\"job\"] = create_job_client()\n", + "\n", + "for client in clients.items():\n", + " print(client)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0ce08bfdc2d0" + }, + "source": [ + "## Prepare a trainer script" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "1e930837e6a2" + }, + "source": [ + "### Package assembly" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "06f649f6992d" + }, + "outputs": [], + "source": [ + "# Make folder for python training script\n", + "! rm -rf custom\n", + "! mkdir custom\n", + "\n", + "# Add package information\n", + "! touch custom/README.md\n", + "\n", + "setup_cfg = \"[egg_info]\\n\\\n", + "tag_build =\\n\\\n", + "tag_date = 0\"\n", + "! echo \"$setup_cfg\" > custom/setup.cfg\n", + "\n", + "setup_py = \"import setuptools\\n\\\n", + "setuptools.setup(\\n\\\n", + " install_requires=[\\n\\\n", + " ],\\n\\\n", + " packages=setuptools.find_packages())\"\n", + "! echo \"$setup_py\" > custom/setup.py\n", + "\n", + "pkg_info = \"Metadata-Version: 1.0\\n\\\n", + "Name: Custom XGBoost Iris\\n\\\n", + "Version: 0.0.0\\n\\\n", + "Summary: Demonstration training script\\n\\\n", + "Home-page: www.google.com\\n\\\n", + "Author: Google\\n\\\n", + "Author-email: aferlitsch@google.com\\n\\\n", + "License: Public\\n\\\n", + "Description: Demo\\n\\\n", + "Platform: Vertex AI\"\n", + "! echo \"$pkg_info\" > custom/PKG-INFO\n", + "\n", + "# Make the training subfolder\n", + "! mkdir custom/trainer\n", + "! touch custom/trainer/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "40d7d11f7de7" + }, + "source": [ + "### Task.py contents" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "8eada5e8b01d" + }, + "outputs": [], + "source": [ + "%%writefile custom/trainer/task.py\n", + "# Single Instance Training for Iris\n", + "\n", + "import datetime\n", + "import os\n", + "import subprocess\n", + "import sys\n", + "import pandas as pd\n", + "import xgboost as xgb\n", + "\n", + "import argparse\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv('AIP_MODEL_DIR'), type=str, help='Model dir.')\n", + "args = parser.parse_args()\n", + "\n", + "# Download data\n", + "iris_data_filename = 'iris_data.csv'\n", + "iris_target_filename = 'iris_target.csv'\n", + "data_dir = 'gs://cloud-samples-data/ai-platform/iris'\n", + "\n", + "# gsutil outputs everything to stderr so we need to divert it to stdout.\n", + "subprocess.check_call(['gsutil', 'cp', os.path.join(data_dir,\n", + " iris_data_filename),\n", + " iris_data_filename], stderr=sys.stdout)\n", + "subprocess.check_call(['gsutil', 'cp', os.path.join(data_dir,\n", + " iris_target_filename),\n", + " iris_target_filename], stderr=sys.stdout)\n", + "\n", + "\n", + "# Load data into pandas, then use `.values` to get NumPy arrays\n", + "iris_data = pd.read_csv(iris_data_filename).values\n", + "iris_target = pd.read_csv(iris_target_filename).values\n", + "\n", + "# Convert one-column 2D array into 1D array for use with XGBoost\n", + "iris_target = iris_target.reshape((iris_target.size,))\n", + "\n", + "\n", + "# Load data into DMatrix object\n", + "dtrain = xgb.DMatrix(iris_data, label=iris_target)\n", + "\n", + "# Train XGBoost model\n", + "bst = xgb.train({}, dtrain, 20)\n", + "\n", + "# Export the classifier to a file\n", + "model_filename = 'model.bst'\n", + "bst.save_model(model_filename)\n", + "\n", + "\n", + "# Upload the saved model file to Cloud Storage\n", + "gcs_model_path = os.path.join(args.model_dir, model_filename)\n", + "subprocess.check_call(['gsutil', 'cp', model_filename, gcs_model_path],\n", + " stderr=sys.stdout)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "3022bce50fbf" + }, + "source": [ + "### Store training script on your Cloud Storage bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "27d6604e4590" + }, + "outputs": [], + "source": [ + "! rm -f custom.tar custom.tar.gz\n", + "! tar cvf custom.tar custom\n", + "! gzip custom.tar\n", + "! gsutil cp custom.tar.gz gs://$BUCKET_NAME/iris.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "text_create_and_deploy_model:migration" + }, + "source": [ + "## Train a model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0oqIBOSnJjkW" + }, + "source": [ + "### [projects.locations.customJobs.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.trainingPipelines/create)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "e110f8131d32" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "e3f2dd83c3f9" + }, + "outputs": [], + "source": [ + "TRAIN_IMAGE = \"gcr.io/cloud-aiplatform/training/xgboost-cpu.1-1:latest\"\n", + "\n", + "JOB_NAME = \"custom_job_XGB\" + TIMESTAMP\n", + "\n", + "WORKER_POOL_SPEC = [\n", + " {\n", + " \"replica_count\": 1,\n", + " \"machine_spec\": {\"machine_type\": \"n1-standard-4\"},\n", + " \"python_package_spec\": {\n", + " \"executor_image_uri\": TRAIN_IMAGE,\n", + " \"package_uris\": [\"gs://\" + BUCKET_NAME + \"/iris.tar.gz\"],\n", + " \"python_module\": \"trainer.task\",\n", + " \"args\": [\"--model-dir=\" + \"gs://{}/{}\".format(BUCKET_NAME, JOB_NAME)],\n", + " },\n", + " }\n", + "]\n", + "\n", + "training_job = aip.CustomJob(\n", + " display_name=JOB_NAME, job_spec={\"worker_pool_specs\": WORKER_POOL_SPEC}\n", + ")\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateCustomJobRequest(parent=PARENT, custom_job=training_job).__dict__[\n", + " \"_pb\"\n", + " ]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"customJob\": {\n", + " \"displayName\": \"custom_job_XGB20210323142337\",\n", + " \"jobSpec\": {\n", + " \"workerPoolSpecs\": [\n", + " {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-4\"\n", + " },\n", + " \"replicaCount\": \"1\",\n", + " \"pythonPackageSpec\": {\n", + " \"executorImageUri\": \"gcr.io/cloud-aiplatform/training/xgboost-cpu.1-1:latest\",\n", + " \"packageUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210323142337/iris.tar.gz\"\n", + " ],\n", + " \"pythonModule\": \"trainer.task\",\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210323142337/custom_job_XGB20210323142337\"\n", + " ]\n", + " }\n", + " }\n", + " ]\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "1fcd4e82a52b" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "fbe59127c6f6" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].create_custom_job(parent=PARENT, custom_job=training_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "3dffa1c62454" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "4086d1e46b00" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/customJobs/7371064379959148544\",\n", + " \"displayName\": \"custom_job_XGB20210323142337\",\n", + " \"jobSpec\": {\n", + " \"workerPoolSpecs\": [\n", + " {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-4\"\n", + " },\n", + " \"replicaCount\": \"1\",\n", + " \"diskSpec\": {\n", + " \"bootDiskType\": \"pd-ssd\",\n", + " \"bootDiskSizeGb\": 100\n", + " },\n", + " \"pythonPackageSpec\": {\n", + " \"executorImageUri\": \"gcr.io/cloud-aiplatform/training/xgboost-cpu.1-1:latest\",\n", + " \"packageUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210323142337/iris.tar.gz\"\n", + " ],\n", + " \"pythonModule\": \"trainer.task\",\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210323142337/custom_job_XGB20210323142337\"\n", + " ]\n", + " }\n", + " }\n", + " ]\n", + " },\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"createTime\": \"2021-03-23T14:23:45.067026Z\",\n", + " \"updateTime\": \"2021-03-23T14:23:45.067026Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "training_pipeline_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the custom training job\n", + "custom_training_id = request.name\n", + "# The short numeric ID for the custom training job\n", + "custom_training_short_id = custom_training_id.split(\"/\")[-1]\n", + "\n", + "print(custom_training_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0oqIBOSnJjkW" + }, + "source": [ + "### [projects.locations.customJobs.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.trainingPipelines/get)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dd8e5e3427d5" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "734b3788cff1" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].get_custom_job(name=custom_training_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "f145335bc684" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "30eef648d2ec" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/customJobs/7371064379959148544\",\n", + " \"displayName\": \"custom_job_XGB20210323142337\",\n", + " \"jobSpec\": {\n", + " \"workerPoolSpecs\": [\n", + " {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-4\"\n", + " },\n", + " \"replicaCount\": \"1\",\n", + " \"diskSpec\": {\n", + " \"bootDiskType\": \"pd-ssd\",\n", + " \"bootDiskSizeGb\": 100\n", + " },\n", + " \"pythonPackageSpec\": {\n", + " \"executorImageUri\": \"gcr.io/cloud-aiplatform/training/xgboost-cpu.1-1:latest\",\n", + " \"packageUris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210323142337/iris.tar.gz\"\n", + " ],\n", + " \"pythonModule\": \"trainer.task\",\n", + " \"args\": [\n", + " \"--model-dir=gs://migration-ucaip-trainingaip-20210323142337/custom_job_XGB20210323142337\"\n", + " ]\n", + " }\n", + " }\n", + " ]\n", + " },\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"createTime\": \"2021-03-23T14:23:45.067026Z\",\n", + " \"updateTime\": \"2021-03-23T14:23:45.067026Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trainingpipelines_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "while True:\n", + " response = clients[\"job\"].get_custom_job(name=custom_training_id)\n", + " if response.state != aip.PipelineState.PIPELINE_STATE_SUCCEEDED:\n", + " print(\"Training job has not completed:\", response.state)\n", + " if response.state == aip.PipelineState.PIPELINE_STATE_FAILED:\n", + " break\n", + " else:\n", + " print(\"Training Time:\", response.end_time - response.start_time)\n", + " break\n", + " time.sleep(60)\n", + "\n", + "# model artifact output directory on Google Cloud Storage\n", + "model_artifact_dir = (\n", + " response.job_spec.worker_pool_specs[0].python_package_spec.args[0].split(\"=\")[-1]\n", + ")\n", + "print(\"artifact location \" + model_artifact_dir)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "576a7e7ce36c" + }, + "source": [ + "## Deploy the model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "COwVZtxhJjkW" + }, + "source": [ + "### [projects.locations.models.upload](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.models/upload)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "fed9fd1f70cf" + }, + "source": [ + "#### Request" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "9fc46ac8f715" + }, + "outputs": [], + "source": [ + "DEPLOY_IMAGE = \"gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-1:latest\"\n", + "\n", + "model = {\n", + " \"display_name\": \"custom_job_XGB\" + TIMESTAMP,\n", + " \"artifact_uri\": model_artifact_dir,\n", + " \"container_spec\": {\"image_uri\": DEPLOY_IMAGE, \"ports\": [{\"container_port\": 8080}]},\n", + "}\n", + "\n", + "print(MessageToJson(aip.UploadModelRequest(parent=PARENT, model=model).__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"model\": {\n", + " \"displayName\": \"custom_job_XGB20210323142337\",\n", + " \"containerSpec\": {\n", + " \"imageUri\": \"gcr.io/cloud-aiplatform/prediction/xgboost-cpu.1-1:latest\",\n", + " \"ports\": [\n", + " {\n", + " \"containerPort\": 8080\n", + " }\n", + " ]\n", + " },\n", + " \"artifactUri\": \"gs://migration-ucaip-trainingaip-20210323142337/custom_job_XGB20210323142337\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "8641495cc6f9" + }, + "source": [ + "#### Call" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "5a967bc5db1b" + }, + "outputs": [], + "source": [ + "request = clients[\"model\"].upload_model(parent=PARENT, model=model)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "2118f3c18d0f" + }, + "source": [ + "#### Response" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "1de1269f8216" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/2093698837704081408\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "1cffd6195a73" + }, + "outputs": [], + "source": [ + "# The full unique ID for the model version\n", + "model_id = result.model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_predictions:migration" + }, + "source": [ + "## Make batch predictions\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_prediction_file:migration,new" + }, + "source": [ + "### Make a batch prediction file\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_items:automl,icn,csv" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "import tensorflow as tf\n", + "\n", + "INSTANCES = [[1.4, 1.3, 5.1, 2.8], [1.5, 1.2, 4.7, 2.4]]\n", + "\n", + "gcs_input_uri = \"gs://\" + BUCKET_NAME + \"/\" + \"test.jsonl\"\n", + "with tf.io.gfile.GFile(gcs_input_uri, \"w\") as f:\n", + " for i in INSTANCES:\n", + " f.write(str(i) + \"\\n\")\n", + "\n", + "! gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "datasets_import:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "[1.4, 1.3, 5.1, 2.8]\n", + "[1.5, 1.2, 4.7, 2.4]\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_create:migration,new" + }, + "source": [ + "### [projects.locations.batchPredictionJobs.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.batchPredictionJobs/create)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "request:migration" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_create:migration,new,request,icn" + }, + "outputs": [], + "source": [ + "model_parameters = Value(\n", + " struct_value=Struct(\n", + " fields={\n", + " \"confidence_threshold\": Value(number_value=0.5),\n", + " \"max_predictions\": Value(number_value=10000.0),\n", + " }\n", + " )\n", + ")\n", + "\n", + "batch_prediction_job = {\n", + " \"display_name\": \"custom_job_XGB\" + TIMESTAMP,\n", + " \"model\": model_id,\n", + " \"input_config\": {\n", + " \"instances_format\": \"jsonl\",\n", + " \"gcs_source\": {\"uris\": [gcs_input_uri]},\n", + " },\n", + " \"model_parameters\": model_parameters,\n", + " \"output_config\": {\n", + " \"predictions_format\": \"jsonl\",\n", + " \"gcs_destination\": {\n", + " \"output_uri_prefix\": \"gs://\" + f\"{BUCKET_NAME}/batch_output/\"\n", + " },\n", + " },\n", + " \"dedicated_resources\": {\n", + " \"machine_spec\": {\"machine_type\": \"n1-standard-2\"},\n", + " \"starting_replica_count\": 1,\n", + " \"max_replica_count\": 1,\n", + " },\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateBatchPredictionJobRequest(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_create:migration,new,request,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"batchPredictionJob\": {\n", + " \"displayName\": \"custom_job_XGB20210323142337\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/2093698837704081408\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210323142337/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"modelParameters\": {\n", + " \"max_predictions\": 10000.0,\n", + " \"confidence_threshold\": 0.5\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210323142337/batch_output/\"\n", + " }\n", + " },\n", + " \"dedicatedResources\": {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-2\"\n", + " },\n", + " \"startingReplicaCount\": 1,\n", + " \"maxReplicaCount\": 1\n", + " }\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "call:migration" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].create_batch_prediction_job(\n", + " parent=PARENT, batch_prediction_job=batch_prediction_job\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "response:migration" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,request" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_create:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/batchPredictionJobs/1415053872761667584\",\n", + " \"displayName\": \"custom_job_XGB20210323142337\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/2093698837704081408\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210323142337/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"modelParameters\": {\n", + " \"confidence_threshold\": 0.5,\n", + " \"max_predictions\": 10000.0\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210323142337/batch_output/\"\n", + " }\n", + " },\n", + " \"dedicatedResources\": {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-2\"\n", + " },\n", + " \"startingReplicaCount\": 1,\n", + " \"maxReplicaCount\": 1\n", + " },\n", + " \"manualBatchTuningParameters\": {},\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"createTime\": \"2021-03-23T14:25:10.582704Z\",\n", + " \"updateTime\": \"2021-03-23T14:25:10.582704Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batch_job_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The fully qualified ID for the batch job\n", + "batch_job_id = request.name\n", + "# The short numeric ID for the batch job\n", + "batch_job_short_id = batch_job_id.split(\"/\")[-1]\n", + "\n", + "print(batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_get:migration,new" + }, + "source": [ + "### [projects.locations.batchPredictionJobs.get](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.batchPredictionJobs/get)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "call:migration" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_get:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"job\"].get_batch_prediction_job(name=batch_job_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "response:migration" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,request" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_get:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/batchPredictionJobs/1415053872761667584\",\n", + " \"displayName\": \"custom_job_XGB20210323142337\",\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/2093698837704081408\",\n", + " \"inputConfig\": {\n", + " \"instancesFormat\": \"jsonl\",\n", + " \"gcsSource\": {\n", + " \"uris\": [\n", + " \"gs://migration-ucaip-trainingaip-20210323142337/test.jsonl\"\n", + " ]\n", + " }\n", + " },\n", + " \"modelParameters\": {\n", + " \"max_predictions\": 10000.0,\n", + " \"confidence_threshold\": 0.5\n", + " },\n", + " \"outputConfig\": {\n", + " \"predictionsFormat\": \"jsonl\",\n", + " \"gcsDestination\": {\n", + " \"outputUriPrefix\": \"gs://migration-ucaip-trainingaip-20210323142337/batch_output/\"\n", + " }\n", + " },\n", + " \"dedicatedResources\": {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-2\"\n", + " },\n", + " \"startingReplicaCount\": 1,\n", + " \"maxReplicaCount\": 1\n", + " },\n", + " \"manualBatchTuningParameters\": {},\n", + " \"state\": \"JOB_STATE_PENDING\",\n", + " \"createTime\": \"2021-03-23T14:25:10.582704Z\",\n", + " \"updateTime\": \"2021-03-23T14:25:10.582704Z\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "batchpredictionjobs_get:migration,new,wait" + }, + "outputs": [], + "source": [ + "def get_latest_predictions(gcs_out_dir):\n", + " \"\"\" Get the latest prediction subfolder using the timestamp in the subfolder name\"\"\"\n", + " folders = !gsutil ls $gcs_out_dir\n", + " latest = \"\"\n", + " for folder in folders:\n", + " subfolder = folder.split(\"/\")[-2]\n", + " if subfolder.startswith(\"prediction-\"):\n", + " if subfolder > latest:\n", + " latest = folder[:-1]\n", + " return latest\n", + "\n", + "\n", + "while True:\n", + " response = clients[\"job\"].get_batch_prediction_job(name=batch_job_id)\n", + " if response.state != aip.JobState.JOB_STATE_SUCCEEDED:\n", + " print(\"The job has not completed:\", response.state)\n", + " if response.state == aip.JobState.JOB_STATE_FAILED:\n", + " break\n", + " else:\n", + " folder = get_latest_predictions(\n", + " response.output_config.gcs_destination.output_uri_prefix\n", + " )\n", + " ! gsutil ls $folder/prediction*\n", + "\n", + " ! gsutil cat -h $folder/prediction*\n", + " break\n", + " time.sleep(60)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "batchpredictionjobs_get:migration,new,wait,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "==> gs://migration-ucaip-trainingaip-20210323142337/batch_output/prediction-custom_job_XGB20210323142337-2021_03_23T07_25_10_544Z/prediction.errors_stats-00000-of-00001 <==\n", + "\n", + "==> gs://migration-ucaip-trainingaip-20210323142337/batch_output/prediction-custom_job_XGB20210323142337-2021_03_23T07_25_10_544Z/prediction.results-00000-of-00001 <==\n", + "{\"instance\": [1.4, 1.3, 5.1, 2.8], \"prediction\": 2.0451931953430176}\n", + "{\"instance\": [1.5, 1.2, 4.7, 2.4], \"prediction\": 1.9618644714355469}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "be2ec9a417b1" + }, + "source": [ + "## Make online predictions" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_create:migration,new" + }, + "source": [ + "### [projects.locations.endpoints.create](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/create)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "request:migration" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_create:migration,new,request" + }, + "outputs": [], + "source": [ + "endpoint = {\"display_name\": \"custom_job_XGB\" + TIMESTAMP}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.CreateEndpointRequest(parent=PARENT, endpoint=endpoint).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_create:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"parent\": \"projects/migration-ucaip-training/locations/us-central1\",\n", + " \"endpoint\": {\n", + " \"displayName\": \"custom_job_XGB20210323142337\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "call:migration" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_create:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"endpoint\"].create_endpoint(parent=PARENT, endpoint=endpoint)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "response:migration" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,response" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_create:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"name\": \"projects/116273516712/locations/us-central1/endpoints/1733903448723685376\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoint_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The full unique ID for the endpoint\n", + "endpoint_id = result.name\n", + "# The short numeric ID for the endpoint\n", + "endpoint_short_id = endpoint_id.split(\"/\")[-1]\n", + "\n", + "print(endpoint_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_deploymodel:migration,new" + }, + "source": [ + "### [projects.locations.endpoints.deployModel](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/deployModel)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "request:migration" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_deploymodel:migration,new,request" + }, + "outputs": [], + "source": [ + "deployed_model = {\n", + " \"model\": model_id,\n", + " \"display_name\": \"custom_job_XGB\" + TIMESTAMP,\n", + " \"dedicated_resources\": {\n", + " \"min_replica_count\": 1,\n", + " \"max_replica_count\": 1,\n", + " \"machine_spec\": {\"machine_type\": \"n1-standard-4\", \"accelerator_count\": 0},\n", + " },\n", + "}\n", + "\n", + "print(\n", + " MessageToJson(\n", + " aip.DeployModelRequest(\n", + " endpoint=endpoint_id,\n", + " deployed_model=deployed_model,\n", + " traffic_split={\"0\": 100},\n", + " ).__dict__[\"_pb\"]\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_deploymodel:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"endpoint\": \"projects/116273516712/locations/us-central1/endpoints/1733903448723685376\",\n", + " \"deployedModel\": {\n", + " \"model\": \"projects/116273516712/locations/us-central1/models/2093698837704081408\",\n", + " \"displayName\": \"custom_job_XGB20210323142337\",\n", + " \"dedicatedResources\": {\n", + " \"machineSpec\": {\n", + " \"machineType\": \"n1-standard-4\"\n", + " },\n", + " \"minReplicaCount\": 1,\n", + " \"maxReplicaCount\": 1\n", + " }\n", + " },\n", + " \"trafficSplit\": {\n", + " \"0\": 100\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "call:migration" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_deploymodel:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"endpoint\"].deploy_model(\n", + " endpoint=endpoint_id, deployed_model=deployed_model, traffic_split={\"0\": 100}\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "response:migration" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,response" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_deploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"deployedModel\": {\n", + " \"id\": \"7407594554280378368\"\n", + " }\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deployed_model_id:migration,new,response" + }, + "outputs": [], + "source": [ + "# The unique ID for the deployed model\n", + "deployed_model_id = result.deployed_model.id\n", + "\n", + "print(deployed_model_id)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_predict:migration,new" + }, + "source": [ + "### [projects.locations.endpoints.predict](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/predict)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "0d0bbb13eea3" + }, + "source": [ + "### Prepare file for online prediction" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "f35289352a3f" + }, + "outputs": [], + "source": [ + "INSTANCES = [[1.4, 1.3, 5.1, 2.8], [1.5, 1.2, 4.7, 2.4]]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "request:migration" + }, + "source": [ + "#### Request\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_predict:migration,new,request,icn" + }, + "outputs": [], + "source": [ + "prediction_request = {\"endpoint\": endpoint_id, \"instances\": INSTANCES}\n", + "\n", + "print(json.dumps(prediction_request, indent=2))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_deploymodel:migration,new,request" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"endpoint\": \"projects/116273516712/locations/us-central1/endpoints/1733903448723685376\",\n", + " \"instances\": [\n", + " [\n", + " 1.4,\n", + " 1.3,\n", + " 5.1,\n", + " 2.8\n", + " ],\n", + " [\n", + " 1.5,\n", + " 1.2,\n", + " 4.7,\n", + " 2.4\n", + " ]\n", + " ]\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "call:migration" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_predict:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"prediction\"].predict(endpoint=endpoint_id, instances=INSTANCES)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "response:migration" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,request" + }, + "outputs": [], + "source": [ + "print(MessageToJson(request.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_predict:migration,new,response,icn" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{\n", + " \"predictions\": [\n", + " 2.045193195343018,\n", + " 1.961864471435547\n", + " ],\n", + " \"deployedModelId\": \"7407594554280378368\"\n", + "}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new" + }, + "source": [ + "### [projects.locations.endpoints.undeployModel](https://cloud.google.com/vertex-ai/docs/reference/rest/v1beta1/projects.locations.endpoints/undeployModel)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "call:migration" + }, + "source": [ + "#### Call\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "endpoints_undeploymodel:migration,new,call" + }, + "outputs": [], + "source": [ + "request = clients[\"endpoint\"].undeploy_model(\n", + " endpoint=endpoint_id, deployed_model_id=deployed_model_id, traffic_split={}\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "response:migration" + }, + "source": [ + "#### Response\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "print:migration,new,response" + }, + "outputs": [], + "source": [ + "result = request.result()\n", + "\n", + "print(MessageToJson(result.__dict__[\"_pb\"]))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "endpoints_undeploymodel:migration,new,response" + }, + "source": [ + "*Example output*:\n", + "```\n", + "{}\n", + "```\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup:migration,new" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup:migration,new" + }, + "outputs": [], + "source": [ + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_pipeline = True\n", + "delete_batchjob = True\n", + "delete_bucket = True\n", + "\n", + "# Delete the model using the Vertex AI fully qualified identifier for the model\n", + "try:\n", + " if delete_model:\n", + " clients[\"model\"].delete_model(name=model_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex AI fully qualified identifier for the endpoint\n", + "try:\n", + " if delete_endpoint:\n", + " clients[\"endpoint\"].delete_endpoint(name=endpoint_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the custom training using the Vertex AI fully qualified identifier for the custome training\n", + "try:\n", + " if custom_training_id:\n", + " clients[\"job\"].delete_custom_job(name=custom_training_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch job using the Vertex AI fully qualified identifier for the batch job\n", + "try:\n", + " if delete_batchjob:\n", + " clients[\"job\"].delete_batch_prediction_job(name=batch_job_id)\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil rm -r gs://$BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "name": "UJ9 unified Custom Training Prebuilt Container XGBoost.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/ml_metadata/sdk-metric-parameter-tracking-for-custom-jobs.ipynb b/notebooks/community/ml_metadata/sdk-metric-parameter-tracking-for-custom-jobs.ipynb new file mode 100644 index 000000000..1fcd2bfd0 --- /dev/null +++ b/notebooks/community/ml_metadata/sdk-metric-parameter-tracking-for-custom-jobs.ipynb @@ -0,0 +1,1038 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ur8xi4C7S06n" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "JAPoU8Sm5E6e" + }, + "source": [ + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "j9gUDU_3vV9d" + }, + "source": [ + "#Vertex AI: Track parameters and metrics for custom training jobs" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tvgnzT1CKxrO" + }, + "source": [ + "## Overview\n", + "\n", + "This notebook demonstrates how to track metrics and parameters for Vertex AI custom training jobs, and how to perform detailed analysis using this data.\n", + "\n", + "### Dataset\n", + "\n", + "This example uses the Abalone Dataset. For more information about this dataset please visit: https://archive.ics.uci.edu/ml/datasets/abalone\n", + "### Objective\n", + "\n", + "In this notebook, you will learn how to use Vertex SDK for Python to:\n", + "\n", + " * Track training parameters and prediction metrics for a custom training job.\n", + " * Extract and perform analysis for all parameters and metrics within an Experiment.\n", + "\n", + "### Costs \n", + "\n", + "\n", + "This tutorial uses billable components of Google Cloud:\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ze4-nDLfK4pw" + }, + "source": [ + "### Set up your local development environment\n", + "\n", + "**If you are using Colab or Google Cloud Notebooks**, your environment already meets\n", + "all the requirements to run this notebook. You can skip this step." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gCuSR8GkAgzl" + }, + "source": [ + "**Otherwise**, make sure your environment meets this notebook's requirements.\n", + "You need the following:\n", + "\n", + "* The Google Cloud SDK\n", + "* Git\n", + "* Python 3\n", + "* virtualenv\n", + "* Jupyter notebook running in a virtual environment with Python 3\n", + "\n", + "The Google Cloud guide to [Setting up a Python development\n", + "environment](https://cloud.google.com/python/setup) and the [Jupyter\n", + "installation guide](https://jupyter.org/install) provide detailed instructions\n", + "for meeting these requirements. The following steps provide a condensed set of\n", + "instructions:\n", + "\n", + "1. [Install and initialize the Cloud SDK.](https://cloud.google.com/sdk/docs/)\n", + "\n", + "1. [Install Python 3.](https://cloud.google.com/python/setup#installing_python)\n", + "\n", + "1. [Install\n", + " virtualenv](https://cloud.google.com/python/setup#installing_and_using_virtualenv)\n", + " and create a virtual environment that uses Python 3. Activate the virtual environment.\n", + "\n", + "1. To install Jupyter, run `pip install jupyter` on the\n", + "command-line in a terminal shell.\n", + "\n", + "1. To launch Jupyter, run `jupyter notebook` on the command-line in a terminal shell.\n", + "\n", + "1. Open this notebook in the Jupyter Notebook Dashboard." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "i7EUnXsZhAGF" + }, + "source": [ + "### Install additional packages\n", + "\n", + "Run the following commands to install the Vertex SDK for Python." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "IaYsrh0Tc17L" + }, + "outputs": [], + "source": [ + "import sys\n", + "\n", + "if \"google.colab\" in sys.modules:\n", + " USER_FLAG = \"\"\n", + "else:\n", + " USER_FLAG = \"--user\"" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "qblyW_dcyOQA" + }, + "outputs": [], + "source": [ + "!python3 -m pip install {USER_FLAG} google-cloud-aiplatform --upgrade" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "hhq5zEbGg0XX" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "After you install the additional packages, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "EzrelQZ22IZj" + }, + "outputs": [], + "source": [ + "# Automatically restart kernel after installs\n", + "import os\n", + "\n", + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "lWEdiXsJg0XY" + }, + "source": [ + "## Before you begin\n", + "\n", + "### Select a GPU runtime\n", + "\n", + "**Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select \"Runtime --> Change runtime type > GPU\"**" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "BF1j6f9HApxa" + }, + "source": [ + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n", + "\n", + "1. [Enable the Vertex AI API and Compute Engine API](https://console.cloud.google.com/flows/enableapi?apiid=aiplatform.googleapis.com,compute_component).\n", + "\n", + "1. If you are running this notebook locally, you will need to install the [Cloud SDK](https://cloud.google.com/sdk).\n", + "\n", + "1. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "WReHDGG5g0XY" + }, + "source": [ + "#### Set your project ID\n", + "\n", + "**If you don't know your project ID**, you may be able to get your project ID using `gcloud`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "oM1iC_MfAts1" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "PROJECT_ID = \"\"\n", + "\n", + "# Get your Google Cloud project ID from gcloud\n", + "if not os.getenv(\"IS_TESTING\"):\n", + " shell_output=!gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID: \", PROJECT_ID)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "qJYoRfYng0XZ" + }, + "source": [ + "Otherwise, set your project ID here." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "riG_qUokg0XZ" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None:\n", + " PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "XsnuGoJM9mUw" + }, + "source": [ + "Set gcloud config to your project ID." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "TL9QIaVd9hvm" + }, + "outputs": [], + "source": [ + "!gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "06571eb4063b" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append it onto the name of resources you create in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "697568e92bd6" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dr--iN2kAylZ" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebooks**, your environment is already\n", + "authenticated. Skip this step." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sBCra4QMA2wR" + }, + "source": [ + "**If you are using Colab**, run the cell below and follow the instructions\n", + "when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "1. In the Cloud Console, go to the [**Create service account key**\n", + " page](https://console.cloud.google.com/apis/credentials/serviceaccountkey).\n", + "\n", + "2. Click **Create service account**.\n", + "\n", + "3. In the **Service account name** field, enter a name, and\n", + " click **Create**.\n", + "\n", + "4. In the **Grant this service account access to project** section, click the **Role** drop-down list. Type \"Vertex AI\"\n", + "into the filter box, and select\n", + " **Vertex AI Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "5. Click *Create*. A JSON file that contains your key downloads to your\n", + "local environment.\n", + "\n", + "6. Enter the path to your service account key as the\n", + "`GOOGLE_APPLICATION_CREDENTIALS` variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "PyQmSRbKA8r-" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebooks, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "zgPO1eR3CYjk" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "\n", + "When you submit a training job using the Cloud SDK, you upload a Python package\n", + "containing your training code to a Cloud Storage bucket. Vertex AI runs\n", + "the code from this package. In this tutorial, Vertex AI also saves the\n", + "trained model that results from your job in the same bucket. Using this model artifact, you can then\n", + "create Vertex AI model and endpoint resources in order to serve\n", + "online predictions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all\n", + "Cloud Storage buckets.\n", + "\n", + "You may also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Make sure to [choose a region where Vertex AI services are\n", + "available](https://cloud.google.com/vertex-ai/docs/general/locations#available_regions). You may\n", + "not use a Multi-Regional Storage bucket for training with Vertex AI." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "MzGDU7TWdts_" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}\n", + "REGION = \"[your-region]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cf221059d072" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"-aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "-EcIXiGsCePi" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "NIq7R4HZCfIc" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ucvCsknMCims" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "vhOb7YnwClBb" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "XoEqT2Y4DJmf" + }, + "source": [ + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Y9Uo3tifg1kx" + }, + "source": [ + "Import required libraries.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pRUOFELefqf1" + }, + "outputs": [], + "source": [ + "import pandas as pd\n", + "from google.cloud import aiplatform\n", + "from sklearn.metrics import mean_absolute_error, mean_squared_error\n", + "from tensorflow.python.keras.utils import data_utils" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "O8XJZB3gR8eL" + }, + "source": [ + "## Initialize Vertex AI and set an _experiment_\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "xtXZWmYqJ1bh" + }, + "source": [ + "Define experiment name." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "JIOrI-hoJ46P" + }, + "outputs": [], + "source": [ + "EXPERIMENT_NAME = \"\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "jWQLXXNVN4Lv" + }, + "source": [ + "If EXEPERIMENT_NAME is not set, set a default one below:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Q1QInYWOKsmo" + }, + "outputs": [], + "source": [ + "if EXPERIMENT_NAME == \"\" or EXPERIMENT_NAME is None:\n", + " EXPERIMENT_NAME = \"my-experiment-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "DKIsYVjj56_X" + }, + "source": [ + "Initialize the *client* for Vertex AI." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Wrlk2B2nJ7-X" + }, + "outputs": [], + "source": [ + "aiplatform.init(\n", + " project=PROJECT_ID,\n", + " location=REGION,\n", + " staging_bucket=BUCKET_NAME,\n", + " experiment=EXPERIMENT_NAME,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "6PlilQPFeS_h" + }, + "source": [ + "## Tracking parameters and metrics in Vertex AI custom training jobs" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "9nokDKBAxwV8" + }, + "source": [ + "This example uses the Abalone Dataset. For more information about this dataset please visit: https://archive.ics.uci.edu/ml/datasets/abalone" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "V_T10yTTqcS_" + }, + "outputs": [], + "source": [ + "!wget https://storage.googleapis.com/download.tensorflow.org/data/abalone_train.csv\n", + "!gsutil cp abalone_train.csv {BUCKET_NAME}/data/\n", + "\n", + "gcs_csv_path = f\"{BUCKET_NAME}/data/abalone_train.csv\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "35QVNhACqcTJ" + }, + "source": [ + "### Create a managed tabular dataset from a CSV\n", + "\n", + "A Managed dataset can be used to create an AutoML model or a custom model. " + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "4OfCqaYRqcTJ" + }, + "outputs": [], + "source": [ + "ds = aiplatform.TabularDataset.create(display_name=\"abalone\", gcs_source=[gcs_csv_path])\n", + "\n", + "ds.resource_name" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "VcEOYYolqcTN" + }, + "source": [ + "### Write the training script\n", + "\n", + "Run the following cell to create the training script that is used in the sample custom training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "OauJqJmJqcTO" + }, + "outputs": [], + "source": [ + "%%writefile training_script.py\n", + "\n", + "import pandas as pd\n", + "import argparse\n", + "import os\n", + "import tensorflow as tf\n", + "from tensorflow import keras\n", + "from tensorflow.keras import layers\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=10, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--num_units', dest='num_units',\n", + " default=64, type=int,\n", + " help='Number of unit for first layer.')\n", + "args = parser.parse_args()\n", + "# uncomment and bump up replica_count for distributed training\n", + "# strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "# tf.distribute.experimental_set_strategy(strategy)\n", + "\n", + "col_names = [\"Length\", \"Diameter\", \"Height\", \"Whole weight\", \"Shucked weight\", \"Viscera weight\", \"Shell weight\", \"Age\"]\n", + "target = \"Age\"\n", + "\n", + "def aip_data_to_dataframe(wild_card_path):\n", + " return pd.concat([pd.read_csv(fp.numpy().decode(), names=col_names)\n", + " for fp in tf.data.Dataset.list_files([wild_card_path])])\n", + "\n", + "def get_features_and_labels(df):\n", + " return df.drop(target, axis=1).values, df[target].values\n", + "\n", + "def data_prep(wild_card_path):\n", + " return get_features_and_labels(aip_data_to_dataframe(wild_card_path))\n", + "\n", + "\n", + "model = tf.keras.Sequential([layers.Dense(args.num_units), layers.Dense(1)])\n", + "model.compile(loss='mse', optimizer='adam')\n", + "\n", + "model.fit(*data_prep(os.environ[\"AIP_TRAINING_DATA_URI\"]),\n", + " epochs=args.epochs ,\n", + " validation_data=data_prep(os.environ[\"AIP_VALIDATION_DATA_URI\"]))\n", + "print(model.evaluate(*data_prep(os.environ[\"AIP_TEST_DATA_URI\"])))\n", + "\n", + "# save as Vertex AI Managed model\n", + "tf.saved_model.save(model, os.environ[\"AIP_MODEL_DIR\"])" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Yp2clkOJSDhR" + }, + "source": [ + "### Launch a custom training job and track its trainig parameters on Vertex AI ML Metadata" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "btb6d48lqcTT" + }, + "outputs": [], + "source": [ + "job = aiplatform.CustomTrainingJob(\n", + " display_name=\"train-abalone-dist-1-replica\",\n", + " script_path=\"training_script.py\",\n", + " container_uri=\"gcr.io/cloud-aiplatform/training/tf-cpu.2-2:latest\",\n", + " requirements=[\"gcsfs==0.7.1\"],\n", + " model_serving_container_image_uri=\"gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest\",\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "k_QorXXztzPH" + }, + "source": [ + "Start a new experiment run to track training parameters and start the training job. Note that this operation will take around 10 mins." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "oVTORjQpJ7-Y" + }, + "outputs": [], + "source": [ + "aiplatform.start_run(\"custom-training-run-1\") # Change this to your desired run name\n", + "parameters = {\"epochs\": 10, \"num_units\": 64}\n", + "aiplatform.log_params(parameters)\n", + "\n", + "model = job.run(\n", + " ds,\n", + " replica_count=1,\n", + " model_display_name=\"abalone-model\",\n", + " args=[f\"--epochs={parameters['epochs']}\", f\"--num_units={parameters['num_units']}\"],\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "5vhDsMJNqcTW" + }, + "source": [ + "### Deploy Model and calculate prediction metrics" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "O-uCOL3Naap4" + }, + "source": [ + "Deploy model to Google Cloud. This operation will take 10-20 mins." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Y9GH72wWqcTX" + }, + "outputs": [], + "source": [ + "endpoint = model.deploy(machine_type=\"n1-standard-4\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "JY-5skFhasWs" + }, + "source": [ + "Once model is deployed, perform online prediction using the `abalone_test` dataset and calculate prediction metrics." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "saw50bqwa-dR" + }, + "source": [ + "Prepare the prediction dataset." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ABZQmqsWISQv" + }, + "outputs": [], + "source": [ + "def read_data(uri):\n", + " dataset_path = data_utils.get_file(\"abalone_test.data\", uri)\n", + " col_names = [\n", + " \"Length\",\n", + " \"Diameter\",\n", + " \"Height\",\n", + " \"Whole weight\",\n", + " \"Shucked weight\",\n", + " \"Viscera weight\",\n", + " \"Shell weight\",\n", + " \"Age\",\n", + " ]\n", + " dataset = pd.read_csv(\n", + " dataset_path,\n", + " names=col_names,\n", + " na_values=\"?\",\n", + " comment=\"\\t\",\n", + " sep=\",\",\n", + " skipinitialspace=True,\n", + " )\n", + " return dataset\n", + "\n", + "\n", + "def get_features_and_labels(df):\n", + " target = \"Age\"\n", + " return df.drop(target, axis=1).values, df[target].values\n", + "\n", + "\n", + "test_dataset, test_labels = get_features_and_labels(\n", + " read_data(\n", + " \"https://storage.googleapis.com/download.tensorflow.org/data/abalone_test.csv\"\n", + " )\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "_HphZ38obJeB" + }, + "source": [ + "Perform online prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "eXD-OvsrKmCt" + }, + "outputs": [], + "source": [ + "prediction = endpoint.predict(test_dataset.tolist())\n", + "prediction" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "TDKiv_O7bNwE" + }, + "source": [ + "Calculate and track prediction evaluation metrics." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cj0fHucbKopn" + }, + "outputs": [], + "source": [ + "mse = mean_squared_error(test_labels, prediction.predictions)\n", + "mae = mean_absolute_error(test_labels, prediction.predictions)\n", + "\n", + "aiplatform.log_metrics({\"mse\": mse, \"mae\": mae})" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "CCGmesdIbbHf" + }, + "source": [ + "### Extract all parameters and metrics created during this experiment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "KlcEBou-Pl4Z" + }, + "outputs": [], + "source": [ + "aiplatform.get_experiment_df()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "WTHvPMweMlP1" + }, + "source": [ + "### View data in the Cloud Console" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "F19_5lw0MqXv" + }, + "source": [ + "Parameters and metrics can also be viewed in the Cloud Console. \n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "GmN9vE9pqqzt" + }, + "outputs": [], + "source": [ + "print(\"Vertex AI Experiments:\")\n", + "print(\n", + " f\"https://console.cloud.google.com/ai/platform/experiments/experiments?folder=&organizationId=&project={PROJECT_ID}\"\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "TpV-iwP9qw9c" + }, + "source": [ + "## Cleaning up\n", + "\n", + "To clean up all Google Cloud resources used in this project, you can [delete the Google Cloud\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "Training Job\n", + "Model\n", + "Cloud Storage Bucket\n", + "\n", + "* Training Job\n", + "* Model\n", + "* Endpoint\n", + "* Cloud Storage Bucket\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "rwPZoZISHhaY" + }, + "outputs": [], + "source": [ + "delete_training_job = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "\n", + "# Warning: Setting this to true will delete everything in your bucket\n", + "delete_bucket = False\n", + "\n", + "# Delete the training job\n", + "job.delete()\n", + "\n", + "# Delete the model\n", + "model.delete()\n", + "\n", + "# Delete the endpoint\n", + "endpoint.delete()\n", + "\n", + "if delete_bucket and \"BUCKET_NAME\" in globals():\n", + " ! gsutil -m rm -r $BUCKET_NAME" + ] + } + ], + "metadata": { + "colab": { + "collapsed_sections": [], + "name": "sdk-metric-parameter-tracking-for-custom-jobs.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/ml_metadata/sdk-metric-parameter-tracking-for-locally-trained-models.ipynb b/notebooks/community/ml_metadata/sdk-metric-parameter-tracking-for-locally-trained-models.ipynb new file mode 100644 index 000000000..fa85cc4bb --- /dev/null +++ b/notebooks/community/ml_metadata/sdk-metric-parameter-tracking-for-locally-trained-models.ipynb @@ -0,0 +1,877 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ur8xi4C7S06n" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "JAPoU8Sm5E6e" + }, + "source": [ + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "WBFL9LagqmwT" + }, + "source": [ + "#Vertex AI: Track parameters and metrics for locally trained models" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tvgnzT1CKxrO" + }, + "source": [ + "## Overview\n", + "\n", + "This notebook demonstrates how to track metrics and parameters for ML training jobs and analyze this metadata using Vertex SDK for Python.\n", + "\n", + "### Dataset\n", + "\n", + "In this notebook, we will train a simple distributed neural network (DNN) model to predict automobile's miles per gallon (MPG) based on automobile information in the [auto-mpg dataset](https://www.kaggle.com/devanshbesain/exploration-and-analysis-auto-mpg).\n", + "\n", + "### Objective\n", + "\n", + "In this notebook, you will learn how to use Vertex SDK for Python to:\n", + "\n", + " * Track parameters and metrics for a locally trainined model.\n", + " * Extract and perform analysis for all parameters and metrics within an Experiment.\n", + "\n", + "### Costs \n", + "\n", + "\n", + "This tutorial uses billable components of Google Cloud:\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ze4-nDLfK4pw" + }, + "source": [ + "### Set up your local development environment\n", + "\n", + "**If you are using Colab or Google Cloud Notebooks**, your environment already meets\n", + "all the requirements to run this notebook. You can skip this step." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gCuSR8GkAgzl" + }, + "source": [ + "**Otherwise**, make sure your environment meets this notebook's requirements.\n", + "You need the following:\n", + "\n", + "* The Google Cloud SDK\n", + "* Git\n", + "* Python 3\n", + "* virtualenv\n", + "* Jupyter notebook running in a virtual environment with Python 3\n", + "\n", + "The Google Cloud guide to [Setting up a Python development\n", + "environment](https://cloud.google.com/python/setup) and the [Jupyter\n", + "installation guide](https://jupyter.org/install) provide detailed instructions\n", + "for meeting these requirements. The following steps provide a condensed set of\n", + "instructions:\n", + "\n", + "1. [Install and initialize the Cloud SDK.](https://cloud.google.com/sdk/docs/)\n", + "\n", + "1. [Install Python 3.](https://cloud.google.com/python/setup#installing_python)\n", + "\n", + "1. [Install\n", + " virtualenv](https://cloud.google.com/python/setup#installing_and_using_virtualenv)\n", + " and create a virtual environment that uses Python 3. Activate the virtual environment.\n", + "\n", + "1. To install Jupyter, run `pip install jupyter` on the\n", + "command-line in a terminal shell.\n", + "\n", + "1. To launch Jupyter, run `jupyter notebook` on the command-line in a terminal shell.\n", + "\n", + "1. Open this notebook in the Jupyter Notebook Dashboard." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "i7EUnXsZhAGF" + }, + "source": [ + "### Install additional packages\n", + "\n", + "Run the following commands to install the Vertex SDK for Python." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "IaYsrh0Tc17L" + }, + "outputs": [], + "source": [ + "import sys\n", + "\n", + "if \"google.colab\" in sys.modules:\n", + " USER_FLAG = \"\"\n", + "else:\n", + " USER_FLAG = \"--user\"" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wyy5Lbnzg5fi" + }, + "outputs": [], + "source": [ + "!python3 -m pip install {USER_FLAG} google-cloud-aiplatform --upgrade" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "hhq5zEbGg0XX" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "After you install the additional packages, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "EzrelQZ22IZj" + }, + "outputs": [], + "source": [ + "# Automatically restart kernel after installs\n", + "import os\n", + "\n", + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "lWEdiXsJg0XY" + }, + "source": [ + "## Before you begin\n", + "\n", + "### Select a GPU runtime\n", + "\n", + "**Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select \"Runtime --> Change runtime type > GPU\"**" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "BF1j6f9HApxa" + }, + "source": [ + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n", + "\n", + "1. [Enable the Vertex AI API](https://console.cloud.google.com/flows/enableapi?apiid=aiplatform.googleapis.com).\n", + "\n", + "1. If you are running this notebook locally, you will need to install the [Cloud SDK](https://cloud.google.com/sdk).\n", + "\n", + "1. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "WReHDGG5g0XY" + }, + "source": [ + "#### Set your project ID\n", + "\n", + "**If you don't know your project ID**, you may be able to get your project ID using `gcloud`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "oM1iC_MfAts1" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "PROJECT_ID = \"\"\n", + "\n", + "# Get your Google Cloud project ID from gcloud\n", + "if not os.getenv(\"IS_TESTING\"):\n", + " shell_output=!gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID: \", PROJECT_ID)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "qJYoRfYng0XZ" + }, + "source": [ + "Otherwise, set your project ID here." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "riG_qUokg0XZ" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None:\n", + " PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "06571eb4063b" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append it onto the name of resources you create in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "697568e92bd6" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dr--iN2kAylZ" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebooks**, your environment is already\n", + "authenticated. Skip this step." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sBCra4QMA2wR" + }, + "source": [ + "**If you are using Colab**, run the cell below and follow the instructions\n", + "when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "1. In the Cloud Console, go to the [**Create service account key**\n", + " page](https://console.cloud.google.com/apis/credentials/serviceaccountkey).\n", + "\n", + "2. Click **Create service account**.\n", + "\n", + "3. In the **Service account name** field, enter a name, and\n", + " click **Create**.\n", + "\n", + "4. In the **Grant this service account access to project** section, click the **Role** drop-down list. Type \"Vertex AI\"\n", + "into the filter box, and select\n", + " **Vertex AI Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "5. Click *Create*. A JSON file that contains your key downloads to your\n", + "local environment.\n", + "\n", + "6. Enter the path to your service account key as the\n", + "`GOOGLE_APPLICATION_CREDENTIALS` variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "PyQmSRbKA8r-" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebooks, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "XoEqT2Y4DJmf" + }, + "source": [ + "### Import libraries and define constants" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Y9Uo3tifg1kx" + }, + "source": [ + "Import required libraries." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "pRUOFELefqf1" + }, + "outputs": [], + "source": [ + "import matplotlib.pyplot as plt\n", + "import pandas as pd\n", + "from google.cloud import aiplatform\n", + "from tensorflow.python.keras import Sequential, layers\n", + "from tensorflow.python.keras.utils import data_utils" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "xtXZWmYqJ1bh" + }, + "source": [ + "Define some constants" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "JIOrI-hoJ46P" + }, + "outputs": [], + "source": [ + "EXPERIMENT_NAME = \"\" # @param {type:\"string\"}\n", + "REGION = \"[your-region]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "jWQLXXNVN4Lv" + }, + "source": [ + "If EXEPERIMENT_NAME is not set, set a default one below:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Q1QInYWOKsmo" + }, + "outputs": [], + "source": [ + "if EXPERIMENT_NAME == \"\" or EXPERIMENT_NAME is None:\n", + " EXPERIMENT_NAME = \"my-experiment-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Xuny18aMcWDb" + }, + "source": [ + "## Concepts\n", + "\n", + "To better understanding how parameters and metrics are stored and organized, we'd like to introduce the following concepts:\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "NThDci5bp0Uw" + }, + "source": [ + "### Experiment\n", + "Experiments describe a context that groups your runs and the artifacts you create into a logical session. For example, in this notebook you create an Experiment and log data to that experiment." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "SAyRR3Ydp4X5" + }, + "source": [ + "### Run\n", + "A run represents a single path/avenue that you executed while performing an experiment. A run includes artifacts that you used as inputs or outputs, and parameters that you used in this execution. An Experiment can contain multiple runs. " + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "l1YW2pgyegFP" + }, + "source": [ + "## Getting started tracking parameters and metrics\n", + "\n", + "You can use the Vertex SDK for Python to track metrics and parameters for models trained locally. \n", + "\n", + "In the following example, you train a simple distributed neural network (DNN) model to predict automobile's miles per gallon (MPG) based on automobile information in the [auto-mpg dataset](https://www.kaggle.com/devanshbesain/exploration-and-analysis-auto-mpg)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "KPY41M9_AhZU" + }, + "source": [ + "### Load and process the training dataset" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bfMQSmRuUuX-" + }, + "source": [ + "Download and process the dataset." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "RiQuMv4bmpuV" + }, + "outputs": [], + "source": [ + "def read_data(uri):\n", + " dataset_path = data_utils.get_file(\"auto-mpg.data\", uri)\n", + " column_names = [\n", + " \"MPG\",\n", + " \"Cylinders\",\n", + " \"Displacement\",\n", + " \"Horsepower\",\n", + " \"Weight\",\n", + " \"Acceleration\",\n", + " \"Model Year\",\n", + " \"Origin\",\n", + " ]\n", + " raw_dataset = pd.read_csv(\n", + " dataset_path,\n", + " names=column_names,\n", + " na_values=\"?\",\n", + " comment=\"\\t\",\n", + " sep=\" \",\n", + " skipinitialspace=True,\n", + " )\n", + " dataset = raw_dataset.dropna()\n", + " dataset[\"Origin\"] = dataset[\"Origin\"].map(\n", + " lambda x: {1: \"USA\", 2: \"Europe\", 3: \"Japan\"}.get(x)\n", + " )\n", + " dataset = pd.get_dummies(dataset, prefix=\"\", prefix_sep=\"\")\n", + " return dataset\n", + "\n", + "\n", + "dataset = read_data(\n", + " \"http://archive.ics.uci.edu/ml/machine-learning-databases/auto-mpg/auto-mpg.data\"\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Y06J7A7yU21t" + }, + "source": [ + "Split dataset for training and testing." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "p5JBCBKyH-NC" + }, + "outputs": [], + "source": [ + "def train_test_split(dataset, split_frac=0.8, random_state=0):\n", + " train_dataset = dataset.sample(frac=split_frac, random_state=random_state)\n", + " test_dataset = dataset.drop(train_dataset.index)\n", + " train_labels = train_dataset.pop(\"MPG\")\n", + " test_labels = test_dataset.pop(\"MPG\")\n", + "\n", + " return train_dataset, test_dataset, train_labels, test_labels\n", + "\n", + "\n", + "train_dataset, test_dataset, train_labels, test_labels = train_test_split(dataset)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gaNNTFPaU7KT" + }, + "source": [ + "Normalize the features in the dataset for better model performance." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "VGq5QCoyIEWJ" + }, + "outputs": [], + "source": [ + "def normalize_dataset(train_dataset, test_dataset):\n", + " train_stats = train_dataset.describe()\n", + " train_stats = train_stats.transpose()\n", + "\n", + " def norm(x):\n", + " return (x - train_stats[\"mean\"]) / train_stats[\"std\"]\n", + "\n", + " normed_train_data = norm(train_dataset)\n", + " normed_test_data = norm(test_dataset)\n", + "\n", + " return normed_train_data, normed_test_data\n", + "\n", + "\n", + "normed_train_data, normed_test_data = normalize_dataset(train_dataset, test_dataset)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "UBXUgxgqA_GB" + }, + "source": [ + "### Define ML model and training function" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "66odBYKrIN4q" + }, + "outputs": [], + "source": [ + "def train(\n", + " train_data,\n", + " train_labels,\n", + " num_units=64,\n", + " activation=\"relu\",\n", + " dropout_rate=0.0,\n", + " validation_split=0.2,\n", + " epochs=1000,\n", + "):\n", + "\n", + " model = Sequential(\n", + " [\n", + " layers.Dense(\n", + " num_units,\n", + " activation=activation,\n", + " input_shape=[len(train_dataset.keys())],\n", + " ),\n", + " layers.Dropout(rate=dropout_rate),\n", + " layers.Dense(num_units, activation=activation),\n", + " layers.Dense(1),\n", + " ]\n", + " )\n", + "\n", + " model.compile(loss=\"mse\", optimizer=\"adam\", metrics=[\"mae\", \"mse\"])\n", + " print(model.summary())\n", + "\n", + " history = model.fit(\n", + " train_data, train_labels, epochs=epochs, validation_split=validation_split\n", + " )\n", + "\n", + " return model, history" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "O8XJZB3gR8eL" + }, + "source": [ + "### Initialize the Model Builder SDK and create an Experiment\n", + "\n", + "Initialize the *client* for Vertex AI and create an experiment." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "o_wnT10RJ7-W" + }, + "outputs": [], + "source": [ + "aiplatform.init(project=PROJECT_ID, location=REGION, experiment=EXPERIMENT_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "u-iTnzt3B6Z_" + }, + "source": [ + "### Start several model training runs\n", + "\n", + "Training parameters and metrics are logged for each run." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "i2wnpu8_7JfV" + }, + "outputs": [], + "source": [ + "parameters = [\n", + " {\"num_units\": 16, \"epochs\": 3, \"dropout_rate\": 0.1},\n", + " {\"num_units\": 16, \"epochs\": 10, \"dropout_rate\": 0.1},\n", + " {\"num_units\": 16, \"epochs\": 10, \"dropout_rate\": 0.2},\n", + " {\"num_units\": 32, \"epochs\": 10, \"dropout_rate\": 0.1},\n", + " {\"num_units\": 32, \"epochs\": 10, \"dropout_rate\": 0.2},\n", + "]\n", + "\n", + "for i, params in enumerate(parameters):\n", + " aiplatform.start_run(run=f\"auto-mpg-local-run-{i}\")\n", + " aiplatform.log_params(params)\n", + " model, history = train(\n", + " normed_train_data,\n", + " train_labels,\n", + " num_units=params[\"num_units\"],\n", + " activation=\"relu\",\n", + " epochs=params[\"epochs\"],\n", + " dropout_rate=params[\"dropout_rate\"],\n", + " )\n", + " aiplatform.log_metrics(\n", + " {metric: values[-1] for metric, values in history.history.items()}\n", + " )\n", + "\n", + " loss, mae, mse = model.evaluate(normed_test_data, test_labels, verbose=2)\n", + " aiplatform.log_metrics({\"eval_loss\": loss, \"eval_mae\": mae, \"eval_mse\": mse})" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "jZLrJZTfL7tE" + }, + "source": [ + "### Extract parameters and metrics into a dataframe for analysis" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "A1PqKxlpOZa2" + }, + "source": [ + "We can also extract all parameters and metrics associated with any Experiment into a dataframe for further analysis." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "jbRf1WoH_vbY" + }, + "outputs": [], + "source": [ + "experiment_df = aiplatform.get_experiment_df()\n", + "experiment_df" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "EYuYgqVCMKU1" + }, + "source": [ + "### Visualizing an experiment's parameters and metrics" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "r8orCj8iJuO1" + }, + "outputs": [], + "source": [ + "plt.rcParams[\"figure.figsize\"] = [15, 5]\n", + "\n", + "ax = pd.plotting.parallel_coordinates(\n", + " experiment_df.reset_index(level=0),\n", + " \"run_name\",\n", + " cols=[\n", + " \"param.num_units\",\n", + " \"param.dropout_rate\",\n", + " \"param.epochs\",\n", + " \"metric.loss\",\n", + " \"metric.val_loss\",\n", + " \"metric.eval_loss\",\n", + " ],\n", + " color=[\"blue\", \"green\", \"pink\", \"red\"],\n", + ")\n", + "ax.set_yscale(\"symlog\")\n", + "ax.legend(bbox_to_anchor=(1.0, 0.5))" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "WTHvPMweMlP1" + }, + "source": [ + "## Visualizing experiments in Cloud Console" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "F19_5lw0MqXv" + }, + "source": [ + "Run the following to get the URL of Vertex AI Experiments for your project.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "GmN9vE9pqqzt" + }, + "outputs": [], + "source": [ + "print(\"Vertex AI Experiments:\")\n", + "print(\n", + " f\"https://console.cloud.google.com/ai/platform/experiments/experiments?folder=&organizationId=&project={PROJECT_ID}\"\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "TpV-iwP9qw9c" + }, + "source": [ + "## Cleaning up\n", + "\n", + "To clean up all Google Cloud resources used in this project, you can [delete the Google Cloud\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial." + ] + } + ], + "metadata": { + "colab": { + "collapsed_sections": [], + "name": "sdk-metric-parameter-tracking-for-locally-trained-models.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/sdk/SDK_AutoML_Forecasting_Model_Training_Example.ipynb b/notebooks/community/sdk/SDK_AutoML_Forecasting_Model_Training_Example.ipynb new file mode 100644 index 000000000..aef94ecd4 --- /dev/null +++ b/notebooks/community/sdk/SDK_AutoML_Forecasting_Model_Training_Example.ipynb @@ -0,0 +1,547 @@ +{ + "nbformat": 4, + "nbformat_minor": 0, + "metadata": { + "colab": { + "name": "Vertex AI SDK - AutoML Forecasting Model Training Example", + "provenance": [], + "collapsed_sections": [] + }, + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.8.6" + } + }, + "cells": [ + { + "cell_type": "code", + "metadata": { + "id": "ur8xi4C7S06n" + }, + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "eHLV0D7Y5jtU" + }, + "source": [ + "# Vertex AI Model Builder SDK: AutoML Forecasting Model Training Example\n", + "\n", + "To use this Colaboratory notebook, you copy the notebook to your own Google Drive and open it with Colaboratory (or Colab). You can run each step, or cell, and see its results. To run a cell, use Shift+Enter. Colab automatically displays the return value of the last line in each cell. For more information about running notebooks in Colab, see the [Colab welcome page](https://colab.research.google.com/notebooks/welcome.ipynb).\n", + "\n", + "This notebook demonstrates how to create an AutoML model based on a time series dataset. It will require you provide a bucket where the dataset will be stored.\n", + "\n", + "Note: you may incur charges for training, prediction, storage or usage of other GCP products in connection with testing this SDK." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "lld3eeJUs5yM" + }, + "source": [ + "# Install Vertex AI SDK, Authenticate, and upload of a Dataset to your GCS bucket\n", + "\n", + "After the SDK installation the kernel will be automatically restarted. You may see this error message `Your session crashed for an unknown reason` which is normal." + ] + }, + { + "cell_type": "code", + "metadata": { + "id": "cMZLb8Arr2AG" + }, + "source": [ + "%%capture\n", + "!pip3 uninstall -y google-cloud-aiplatform\n", + "!pip3 install google-cloud-aiplatform\n", + " \n", + "import IPython\n", + " \n", + "app = IPython.Application.instance()\n", + "app.kernel.do_shutdown(True)" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "code", + "metadata": { + "id": "ApsLDJjdsGPN" + }, + "source": [ + "import sys\n", + "\n", + "if \"google.colab\" in sys.modules:\n", + " from google.colab import auth\n", + "\n", + " auth.authenticate_user()" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "c0SNmTBeD2nV" + }, + "source": [ + "### Enter your project and GCS bucket\n", + "\n", + "Enter your Project ID in the cell below. Then run the cell to make sure the Cloud SDK uses the right project for all the commands in this notebook." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "s19AzYSGLIb9" + }, + "source": [ + "**If you don't know your project ID**, you may be able to get your project ID using gcloud." + ] + }, + { + "cell_type": "code", + "metadata": { + "id": "nwlVqT6RKxG7" + }, + "source": [ + "import os\n", + "\n", + "PROJECT_ID = \"\"\n", + "\n", + "# Get your Google Cloud project ID from gcloud\n", + "if not os.getenv(\"IS_TESTING\"):\n", + " shell_output=!gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID: \", PROJECT_ID)" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "H5E8VB3jLOFC" + }, + "source": [ + "Otherwise, set your project ID here." + ] + }, + { + "cell_type": "code", + "metadata": { + "id": "DrED76XTK9OB" + }, + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None:\n", + " PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "zkJk7agzT6F9" + }, + "source": [ + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append it onto the name of resources you create in this tutorial." + ] + }, + { + "cell_type": "code", + "metadata": { + "id": "qcRkdZBaUAz4" + }, + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "TFfpJs3DQsfo" + }, + "source": [ + "Set the name of your Cloud Storage bucket below. It must be unique across all Cloud Storage buckets.\n", + "\n", + "You may also change the REGION variable, which is used for operations throughout the rest of this notebook. Make sure to [choose a region where Vertex AI services are available](https://cloud.google.com/vertex-ai/docs/general/locations#available_regions). You may not use a Multi-Regional Storage bucket for training with Vertex AI." + ] + }, + { + "cell_type": "code", + "metadata": { + "id": "iqSQT6Z6bekX" + }, + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}\n", + "REGION = \"[your-region]\" # @param {type:\"string\"}" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "code", + "metadata": { + "id": "ukGsLjm-Ki14" + }, + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "-6AQjKlnx0mf" + }, + "source": [ + "The datasets we are using are samples from the [Iowa Liquor Retail Sales](https://pantheon.corp.google.com/marketplace/product/iowa-department-of-commerce/iowa-liquor-sales) dataset. The training sample contains the sales from 2020 and the prediction sample (used in the batch prediction step) contains the January - April sales from 2021." + ] + }, + { + "cell_type": "code", + "metadata": { + "id": "V_T10yTTqcS_" + }, + "source": [ + "TRAINING_DATASET_BQ_PATH = 'bq://bigquery-public-data:iowa_liquor_sales_forecasting.2020_sales_train'" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "rk43VP_IqcTE" + }, + "source": [ + "# Initialize Vertex AI SDK\n", + "\n", + "Initialize the *client* for Vertex AI." + ] + }, + { + "cell_type": "code", + "metadata": { + "id": "VCiC9gBWqcTF" + }, + "source": [ + "from google.cloud import aiplatform\n", + "\n", + "aiplatform.init(project=PROJECT_ID, staging_bucket=BUCKET_NAME)" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "35QVNhACqcTJ" + }, + "source": [ + "# Create a Managed Time Series Dataset from BigQuery\n", + "\n", + "This section will create a dataset from a BigQuery table." + ] + }, + { + "cell_type": "code", + "metadata": { + "id": "4OfCqaYRqcTJ" + }, + "source": [ + "ds = aiplatform.datasets.TimeSeriesDataset.create(\n", + " display_name='iowa_liquor_sales_train',\n", + " bq_source=[TRAINING_DATASET_BQ_PATH])\n", + "\n", + "ds.resource_name" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "6-bBqipfqcTS" + }, + "source": [ + "# Launch a Training Job to Create a Model\n", + "\n", + "Once we have defined your training script, we will create a model." + ] + }, + { + "cell_type": "code", + "metadata": { + "id": "aA41rT_mb-rV" + }, + "source": [ + "time_column = \"date\"\n", + "time_series_identifier_column=\"store_name\"\n", + "target_column=\"sale_dollars\"\n", + "\n", + "job = aiplatform.AutoMLForecastingTrainingJob(\n", + " display_name='train-iowa-liquor-sales-automl_1',\n", + " optimization_objective='minimize-rmse', \n", + " column_transformations=[\n", + " {\"timestamp\": {\"column_name\": time_column}},\n", + " {\"numeric\": {\"column_name\": target_column}},\n", + " {\"categorical\": {\"column_name\": \"city\"}},\n", + " {\"categorical\": {\"column_name\": \"zip_code\"}},\n", + " {\"categorical\": {\"column_name\": \"county\"}},\n", + " ]\n", + ")\n", + "\n", + "# This will take around an hour to run\n", + "model = job.run(\n", + " dataset=ds,\n", + " target_column=target_column,\n", + " time_column=time_column,\n", + " time_series_identifier_column=time_series_identifier_column,\n", + " available_at_forecast_columns=[time_column],\n", + " unavailable_at_forecast_columns=[target_column],\n", + " time_series_attribute_columns=[\"city\", \"zip_code\", \"county\"],\n", + " forecast_horizon=30,\n", + " context_window=30,\n", + " data_granularity_unit=\"day\",\n", + " data_granularity_count=1,\n", + " weight_column=None,\n", + " budget_milli_node_hours=1000,\n", + " model_display_name=\"iowa-liquor-sales-forecast-model\", \n", + " predefined_split_column_name=None,\n", + ")" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "code", + "metadata": { + "id": "muSC-mvgHno7", + "cellView": "form" + }, + "source": [ + "#@title # Fetch Model Evaluation Metrics\n", + "#@markdown Fetch the model evaluation metrics calculated during training on the test set.\n", + "\n", + "import pandas as pd\n", + "\n", + "list_evaluation_pager = model.api_client.list_model_evaluations(parent=model.resource_name)\n", + "for model_evaluation in list_evaluation_pager:\n", + " metrics_dict = {m[0]: m[1] for m in model_evaluation.metrics.items()}\n", + " df = pd.DataFrame(metrics_dict.items(), columns=[\"Metric\", \"Value\"])\n", + " print(df.to_string(index=False))" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "nIw1ifPuqcTb" + }, + "source": [ + "# Run Batch Prediction" + ] + }, + { + "cell_type": "code", + "metadata": { + "id": "nT-bZ1autijD" + }, + "source": [ + "#@markdown ## Create Output BigQuery Dataset\n", + "#@markdown First, create a new BigQuery dataset for the batch prediction output in the same region as the batch prediction input dataset. \n", + "\n", + "import os\n", + "from google.cloud import bigquery\n", + "\n", + "os.environ[\"GOOGLE_CLOUD_PROJECT\"] = PROJECT_ID\n", + "\n", + "batch_predict_bq_input_uri = \"bq://bigquery-public-data.iowa_liquor_sales_forecasting.2021_sales_predict\"\n", + "batch_predict_bq_output_dataset_name = \"iowa_liquor_sales_predictions\"\n", + "batch_predict_bq_output_dataset_path = \"{}.{}\".format(PROJECT_ID, batch_predict_bq_output_dataset_name)\n", + "batch_predict_bq_output_uri_prefix = \"bq://{}.{}\".format(PROJECT_ID, batch_predict_bq_output_dataset_name)\n", + "# Must be the same region as batch_predict_bq_input_uri\n", + "client = bigquery.Client()\n", + "dataset = bigquery.Dataset(batch_predict_bq_output_dataset_path)\n", + "dataset_region = \"US\" # @param {type : \"string\"}\n", + "dataset.location = dataset_region\n", + "dataset = client.create_dataset(dataset)\n", + "print(\"Created bigquery dataset {} in {}\".format(batch_predict_bq_output_dataset_path, dataset_region))" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "krKRn9W0xxI2" + }, + "source": [ + "Run a batch prediction job to generate liquor sales forecasts for stores in Iowa from an input dataset containing historical sales." + ] + }, + { + "cell_type": "code", + "metadata": { + "id": "8I8aRjRh6GGG" + }, + "source": [ + "model.batch_predict(\n", + " bigquery_source=batch_predict_bq_input_uri,\n", + " instances_format=\"bigquery\",\n", + " bigquery_destination_prefix=batch_predict_bq_output_uri_prefix,\n", + " predictions_format=\"bigquery\",\n", + " job_display_name=\"predict-iowa-liquor-sales-automl_1\")" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "code", + "metadata": { + "id": "CTQl3fH6Ur2Z", + "cellView": "form" + }, + "source": [ + "#@title # Visualize the Forecasts\n", + "#@markdown Follow the given link to visualize the generated forecasts in [Data Studio](https://support.google.com/datastudio/answer/6283323?hl=en).\n", + "\n", + "import urllib\n", + "\n", + "tables = client.list_tables(batch_predict_bq_output_dataset_path)\n", + "\n", + "prediction_table_id = \"\"\n", + "for table in tables:\n", + " if table.table_id.startswith(\n", + " \"predictions_\") and table.table_id > prediction_table_id:\n", + " prediction_table_id = table.table_id\n", + "batch_predict_bq_output_uri = \"{}.{}\".format(\n", + " batch_predict_bq_output_dataset_path, prediction_table_id)\n", + "\n", + "\n", + "def _sanitize_bq_uri(bq_uri):\n", + " if bq_uri.startswith(\"bq://\"):\n", + " bq_uri = bq_uri[5:]\n", + " return bq_uri.replace(\":\", \".\")\n", + "\n", + "\n", + "def get_data_studio_link(batch_prediction_bq_input_uri,\n", + " batch_prediction_bq_output_uri, time_column,\n", + " time_series_identifier_column, target_column):\n", + " batch_prediction_bq_input_uri = _sanitize_bq_uri(\n", + " batch_prediction_bq_input_uri)\n", + " batch_prediction_bq_output_uri = _sanitize_bq_uri(\n", + " batch_prediction_bq_output_uri)\n", + " base_url = \"https://datastudio.google.com/c/u/0/reporting\"\n", + " query = \"SELECT \\\\n\" \\\n", + " \" CAST(input.{} as DATETIME) timestamp_col,\\\\n\" \\\n", + " \" CAST(input.{} as STRING) time_series_identifier_col,\\\\n\" \\\n", + " \" CAST(input.{} as NUMERIC) historical_values,\\\\n\" \\\n", + " \" CAST(predicted_{}.value as NUMERIC) predicted_values,\\\\n\" \\\n", + " \" * \\\\n\" \\\n", + " \"FROM `{}` input\\\\n\" \\\n", + " \"LEFT JOIN `{}` output\\\\n\" \\\n", + " \"ON\\\\n\" \\\n", + " \"CAST(input.{} as DATETIME) = CAST(output.{} as DATETIME)\\\\n\" \\\n", + " \"AND CAST(input.{} as STRING) = CAST(output.{} as STRING)\"\n", + " query = query.format(time_column, time_series_identifier_column,\n", + " target_column, target_column,\n", + " batch_prediction_bq_input_uri,\n", + " batch_prediction_bq_output_uri, time_column, time_column,\n", + " time_series_identifier_column,\n", + " time_series_identifier_column)\n", + " params = {\n", + " \"templateId\": \"067f70d2-8cd6-4a4c-a099-292acd1053e8\",\n", + " \"ds0.connector\": \"BIG_QUERY\",\n", + " \"ds0.projectId\": PROJECT_ID,\n", + " \"ds0.billingProjectId\": PROJECT_ID,\n", + " \"ds0.type\": \"CUSTOM_QUERY\",\n", + " \"ds0.sql\": query\n", + " }\n", + " params_str_parts = []\n", + " for k, v in params.items():\n", + " params_str_parts.append(\"\\\"{}\\\":\\\"{}\\\"\".format(k, v))\n", + " params_str = \"\".join([\"{\", \",\".join(params_str_parts), \"}\"])\n", + " return \"{}?{}\".format(base_url,\n", + " urllib.parse.urlencode({\"params\": params_str}))\n", + "\n", + "\n", + "print(\n", + " get_data_studio_link(batch_predict_bq_input_uri,\n", + " batch_predict_bq_output_uri, time_column,\n", + " time_series_identifier_column, target_column))" + ], + "execution_count": null, + "outputs": [] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "24NPJ7nCRchZ" + }, + "source": [ + "\n", + "# Cleaning up\n", + "\n", + "To clean up all Google Cloud resources used in this project, you can [delete the Google Cloud project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n" + ] + }, + { + "cell_type": "code", + "metadata": { + "id": "gq3ZSsAkRnXh" + }, + "source": [ + "# Delete model resource\n", + "model.delete(sync=True)\n", + "\n", + "# Delete Cloud Storage objects that were created\n", + "! gsutil -m rm -r $BUCKET_NAME" + ], + "execution_count": null, + "outputs": [] + } + ] +} \ No newline at end of file diff --git a/notebooks/community/sdk/SDK_AutoML_Image_Classification_Training_with_CMEK.ipynb b/notebooks/community/sdk/SDK_AutoML_Image_Classification_Training_with_CMEK.ipynb new file mode 100644 index 000000000..3259d84eb --- /dev/null +++ b/notebooks/community/sdk/SDK_AutoML_Image_Classification_Training_with_CMEK.ipynb @@ -0,0 +1,522 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "VBOfRw7ifk8w" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "w3LR4Lj8fk8x" + }, + "source": [ + "# Feedback or issues?\n", + "For any feedback or questions, please open an [issue](https://github.com/googleapis/python-aiplatform/issues)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "mUMzY5W9fk8y" + }, + "source": [ + "# Vertex SDK for Python: AutoML Image Classfication Training with Customer Managed Encryption Keys (CMEK) Example\n", + "To use this Jupyter notebook, create a copy of the notebook in Colab and open it. You can run each step, or cell, and see its results. To run a cell, use Shift+Enter. Colab automatically displays the return value of the last line in each cell.\n", + "\n", + "This notebook demonstrate how to train an AutoML Image Classification model with CMEK. It will require you provide a bucket where the dataset will be stored.\n", + "\n", + "Note: you may incur charges for training, prediction, storage or usage of other GCP products in connection with testing this SDK." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "lld3eeJUs5yM" + }, + "source": [ + "# Install SDK\n", + " \n", + "After the SDK installation the kernel will be automatically restarted." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "sBfZtR4X1Dr_" + }, + "outputs": [], + "source": [ + "!pip3 uninstall -y google-cloud-aiplatform\n", + "!pip3 install --upgrade google-cloud-kms\n", + "!pip3 install google-cloud-aiplatform\n", + "import IPython\n", + "\n", + "app = IPython.Application.instance()\n", + "app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "c0SNmTBeD2nV" + }, + "source": [ + "### Enter your project and GCS bucket\n", + "\n", + "Enter your Project Id in the cell below. Then run the cell to make sure the Cloud SDK uses the right project for all the commands in this notebook." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "YcwsEwXPivBZ" + }, + "outputs": [], + "source": [ + "import sys\n", + "\n", + "if \"google.colab\" in sys.modules:\n", + " from google.colab import auth\n", + "\n", + " auth.authenticate_user()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "iqSQT6Z6bekX" + }, + "outputs": [], + "source": [ + "REGION = \"YOUR REGION\" # e.g. us-central1\n", + "MY_PROJECT = \"YOUR PROJECT ID\"\n", + "MY_STAGING_BUCKET = \"gs://YOUR BUCKET\" # bucket should be in same region as ucaip" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "mRk9eoTm6Pyi" + }, + "source": [ + "## Setting up Customer Managed Encryption Keys\n", + "\n", + "By default, Google Cloud automatically encrypts data when it is at rest using encryption keys managed by Google. If you have specific compliance or regulatory requirements related to the keys that protect your data, you can use customer-managed encryption keys (CMEK) for your training jobs.\n", + "\n", + "For more info on using CMEK on Vertex AI, please see: [https://cloud.google.com/vertex-ai/docs/general/cmek#before_you_begin](https://cloud.google.com/vertex-ai/docs/general/cmek#before_you_begin)\n", + "\n", + "You can create a key using the guide above or executing the the Notebook cells below." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "RD_Pvrg584X3" + }, + "source": [ + "1. Register your application for Cloud Key Management Service (KMS) API in Google Cloud Platform at https://console.cloud.google.com/flows/enableapi?apiid=cloudkms.googleapis.com\n", + "\n", + "2. Create a key ring\n", + "\n", + "Create a key ring" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dxRZzbvQnZC7" + }, + "outputs": [], + "source": [ + "KEY_RING_ID = \"your_key_ring_name\"\n", + "\n", + "\n", + "# Reference: https://cloud.google.com/kms/docs/samples/kms-create-key-ring\n", + "def create_key_ring(project_id, location_id, id):\n", + " \"\"\"\n", + " Creates a new key ring in Cloud KMS\n", + "\n", + " Args:\n", + " project_id (string): Google Cloud project ID (e.g. 'my-project').\n", + " location_id (string): Cloud KMS location (e.g. 'us-east1').\n", + " id (string): ID of the key ring to create (e.g. 'my-key-ring').\n", + "\n", + " Returns:\n", + " KeyRing: Cloud KMS key ring.\n", + "\n", + " \"\"\"\n", + "\n", + " # Import the client library.\n", + " from google.cloud import kms\n", + "\n", + " # Create the client.\n", + " client = kms.KeyManagementServiceClient()\n", + "\n", + " # Build the parent location name.\n", + " location_name = f\"projects/{project_id}/locations/{location_id}\"\n", + "\n", + " # Build the key ring.\n", + " key_ring = {}\n", + "\n", + " # Call the API.\n", + " created_key_ring = client.create_key_ring(\n", + " request={\"parent\": location_name, \"key_ring_id\": id, \"key_ring\": key_ring}\n", + " )\n", + " print(\"Created key ring: {}\".format(created_key_ring.name))\n", + " return created_key_ring\n", + "\n", + "\n", + "create_key_ring(project_id=MY_PROJECT, location_id=REGION, id=KEY_RING_ID)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gCL1-IfFtWXl" + }, + "source": [ + "Create a key" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "LXcagdmSnYYW" + }, + "outputs": [], + "source": [ + "KEY_ID = \"your_key_id\"\n", + "\n", + "\n", + "# Reference: https://cloud.google.com/kms/docs/samples/kms-create-key-symmetric-encrypt-decrypt\n", + "def create_key_symmetric_encrypt_decrypt(project_id, location_id, key_ring_id, id):\n", + " \"\"\"\n", + " Creates a new symmetric encryption/decryption key in Cloud KMS.\n", + "\n", + " Args:\n", + " project_id (string): Google Cloud project ID (e.g. 'my-project').\n", + " location_id (string): Cloud KMS location (e.g. 'us-east1').\n", + " key_ring_id (string): ID of the Cloud KMS key ring (e.g. 'my-key-ring').\n", + " id (string): ID of the key to create (e.g. 'my-symmetric-key').\n", + "\n", + " Returns:\n", + " CryptoKey: Cloud KMS key.\n", + "\n", + " \"\"\"\n", + "\n", + " # Import the client library.\n", + " from google.cloud import kms\n", + "\n", + " # Create the client.\n", + " client = kms.KeyManagementServiceClient()\n", + "\n", + " # Build the parent key ring name.\n", + " key_ring_name = client.key_ring_path(project_id, location_id, key_ring_id)\n", + "\n", + " # Build the key.\n", + " purpose = kms.CryptoKey.CryptoKeyPurpose.ENCRYPT_DECRYPT\n", + " algorithm = (\n", + " kms.CryptoKeyVersion.CryptoKeyVersionAlgorithm.GOOGLE_SYMMETRIC_ENCRYPTION\n", + " )\n", + " key = {\n", + " \"purpose\": purpose,\n", + " \"version_template\": {\n", + " \"algorithm\": algorithm,\n", + " },\n", + " }\n", + "\n", + " # Call the API.\n", + " created_key = client.create_crypto_key(\n", + " request={\"parent\": key_ring_name, \"crypto_key_id\": id, \"crypto_key\": key}\n", + " )\n", + " print(\"Created symmetric key: {}\".format(created_key.name))\n", + " return created_key\n", + "\n", + "\n", + "create_key_symmetric_encrypt_decrypt(\n", + " project_id=MY_PROJECT, location_id=REGION, key_ring_id=KEY_RING_ID, id=KEY_ID\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "3gKDBOqC8Gl5" + }, + "source": [ + "Give permissions to key to the Vertex AI service account" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "6QrRg08Vqfru" + }, + "outputs": [], + "source": [ + "# Reference: https://cloud.google.com/vertex-ai/docs/general/cmek#granting_permissions\n", + "# Get the service account\n", + "SERVICE_ACCOUNT = ! gcloud projects get-iam-policy {MY_PROJECT} \\\n", + " --flatten=\"bindings[].members\" \\\n", + " --format=\"table(bindings.members)\" \\\n", + " --filter=\"bindings.role:roles/aiplatform.serviceAgent\" \\\n", + " | grep -oP \"service-.+?@gcp-sa-aiplatform.iam.gserviceaccount.com\"\n", + "SERVICE_ACCOUNT = SERVICE_ACCOUNT[0]\n", + "\n", + "print(f\"Service account is: {SERVICE_ACCOUNT}\")\n", + "\n", + "# Give permissions\n", + "!gcloud kms keys add-iam-policy-binding {KEY_ID} \\\n", + " --keyring={KEY_RING_ID} \\\n", + " --location={REGION} \\\n", + " --project={MY_PROJECT} \\\n", + " --member=serviceAccount:{SERVICE_ACCOUNT} \\\n", + " --role=roles/cloudkms.cryptoKeyEncrypterDecrypter" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ebAHZg2vlhXL" + }, + "outputs": [], + "source": [ + "# Create the full resource identifier for the created key\n", + "ENCRYPTION_SPEC_KEY_NAME = f\"projects/{MY_PROJECT}/locations/{REGION}/keyRings/{KEY_RING_ID}/cryptoKeys/{KEY_ID}\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Aa_8wrqSkamz" + }, + "source": [ + "## Initialize Vertex SDK for Python\n", + "\n", + "Initialize the *client* for Vertex AI\n", + "\n", + "All resources created during this Notebook run will encrypted with the encryption key created above.\n", + "\n", + "You can override the encryption key at each function call." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ohdgOs69kGNU" + }, + "outputs": [], + "source": [ + "from google.cloud import aiplatform\n", + "\n", + "aiplatform.init(\n", + " project=MY_PROJECT,\n", + " staging_bucket=MY_STAGING_BUCKET,\n", + " location=REGION,\n", + " encryption_spec_key_name=ENCRYPTION_SPEC_KEY_NAME,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "35QVNhACqcTJ" + }, + "source": [ + "# Create Managed Image Dataset from CSV\n", + "\n", + "This section will create a managed Image dataset from the Flowers dataset. For more imformation on this dataset please visit https://www.tensorflow.org/datasets/catalog/tf_flowers." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "4OfCqaYRqcTJ" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = (\n", + " \"gs://cloud-samples-data/vision/automl_classification/flowers/all_data_v2.csv\"\n", + ")\n", + "\n", + "ds = aiplatform.ImageDataset.create(\n", + " display_name=\"flowers\",\n", + " gcs_source=[IMPORT_FILE],\n", + " import_schema_uri=aiplatform.schema.dataset.ioformat.image.single_label_classification,\n", + ")\n", + "\n", + "ds.resource_name" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "6-bBqipfqcTS" + }, + "source": [ + "# Launch a Training Job to Create a Model\n", + "\n", + "Train an AutoML Image Classification model." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aA41rT_mb-rV" + }, + "outputs": [], + "source": [ + "job = aiplatform.AutoMLImageTrainingJob(\n", + " display_name=\"train-iris-automl-mbsdk-1\",\n", + " prediction_type=\"classification\",\n", + " multi_label=False,\n", + " model_type=\"CLOUD\",\n", + " base_model=None,\n", + ")\n", + "\n", + "# This will take around half an hour to run\n", + "model = job.run(\n", + " dataset=ds,\n", + " model_display_name=\"iris-classification-model-mbsdk\",\n", + " training_fraction_split=0.6,\n", + " validation_fraction_split=0.2,\n", + " test_fraction_split=0.2,\n", + " budget_milli_node_hours=8000,\n", + " disable_early_stopping=False,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "5vhDsMJNqcTW" + }, + "source": [ + "# Deploy Your Model\n", + "\n", + "Deploy your model, then wait until the model FINISHES deployment before proceeding to prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Y9GH72wWqcTX" + }, + "outputs": [], + "source": [ + "endpoint = model.deploy()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "nIw1ifPuqcTb" + }, + "source": [ + "# Predict on Endpoint\n", + "- Take one sample from the data imported to the dataset\n", + "- This sample will be encoded to base64 and passed to the endpoint for prediction" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "H23ISHdHVIZM" + }, + "outputs": [], + "source": [ + "test_item = !gsutil cat $IMPORT_FILE | head -n1\n", + "test_item, test_label = str(test_item[0]).split(\",\")\n", + "\n", + "print(test_item, test_label)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "TF_N0kqZU768" + }, + "outputs": [], + "source": [ + "import base64\n", + "\n", + "import tensorflow as tf\n", + "\n", + "with tf.io.gfile.GFile(test_item, \"rb\") as f:\n", + " content = f.read()\n", + "\n", + "# The format of each instance should conform to the deployed model's prediction input schema.\n", + "instances_list = [{\"content\": base64.b64encode(content).decode(\"utf-8\")}]\n", + "\n", + "prediction = endpoint.predict(instances=instances_list)\n", + "\n", + "prediction" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "nWA3qocXfk82" + }, + "source": [ + "# Undeploy Model from Endpoint" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "V1brMaO_fk82" + }, + "outputs": [], + "source": [ + "endpoint.undeploy_all()" + ] + } + ], + "metadata": { + "colab": { + "collapsed_sections": [], + "name": "AI_Platform_(Unified)_SDK_AutoML_Image_Classification_Training_with_CMEK.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/sdk/SDK_AutoML_Text_Extraction_Training.ipynb b/notebooks/community/sdk/SDK_AutoML_Text_Extraction_Training.ipynb new file mode 100644 index 000000000..3608ac857 --- /dev/null +++ b/notebooks/community/sdk/SDK_AutoML_Text_Extraction_Training.ipynb @@ -0,0 +1,313 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bguORR-uVtyk" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "f1KuZ_LBcHee" + }, + "source": [ + "# Feedback or issues?\n", + "\n", + "For any feedback or questions, please open an [issue](https://github.com/googleapis/python-aiplatform/issues)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "A4QHhG05cJD9" + }, + "source": [ + "# Vertex SDK for Python: AutoML Text Extraction Example\n", + "To use this Jupyter notebook, copy the notebook to a Google Cloud Notebooks instance with Tensorflow installed and open it. You can run each step, or cell, and see its results. To run a cell, use Shift+Enter. Jupyter automatically displays the return value of the last line in each cell. For more information about running notebooks in Google Cloud Notebook, see the [Google Cloud Notebook guide](https://cloud.google.com/vertex-ai/docs/general/notebooks).\n", + "\n", + "\n", + "This notebook demonstrate how to create an AutoML Text Extraction Model, with a Vertex AI text dataset, and how to serve the model for online prediction.\n", + "\n", + "Note: you may incur charges for training, prediction, storage or usage of other GCP products in connection with testing this SDK." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "lld3eeJUs5yM" + }, + "source": [ + "### Install Vertex SDK for Python\n", + "\n", + "\n", + "After the SDK installation the kernel will be automatically restarted." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "sBfZtR4X1Dr_" + }, + "outputs": [], + "source": [ + "!pip3 uninstall -y google-cloud-aiplatform\n", + "!pip3 install google-cloud-aiplatform\n", + "import IPython\n", + "\n", + "app = IPython.Application.instance()\n", + "app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "kBFvlCFh5Yij" + }, + "outputs": [], + "source": [ + "import sys\n", + "\n", + "if \"google.colab\" in sys.modules:\n", + " from google.colab import auth\n", + "\n", + " auth.authenticate_user()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Yz_rkDIteP5M" + }, + "source": [ + "### Enter Your Project and GCS Bucket\n", + "\n", + "Enter your Project Id in the cell below. Then run the cell to make sure the Cloud SDK uses the right project for all the commands in this notebook." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "iqSQT6Z6bekX" + }, + "outputs": [], + "source": [ + "MY_PROJECT = \"YOUR PROJECT ID\"\n", + "MY_STAGING_BUCKET = \"gs://YOUR BUCKET\" # bucket should be in same region as ucaip" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "rk43VP_IqcTE" + }, + "source": [ + "## Initialize Vertex SDK for Python\n", + "\n", + "Initialize the *client* for Vertex AI." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "VCiC9gBWqcTF" + }, + "outputs": [], + "source": [ + "from google.cloud import aiplatform\n", + "\n", + "aiplatform.init(project=MY_PROJECT, staging_bucket=MY_STAGING_BUCKET)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "35QVNhACqcTJ" + }, + "source": [ + "## Create a Dataset on Vertex AI\n", + "We will now create a Vertex AI text dataset using the previously prepared jsonl files. " + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "XGKZ3bdyTMcd" + }, + "source": [ + "### The biomedical dataset\n", + "To create an entity extraction model, use a corpus of biomedical research abstracts that mention hundreds of diseases and concepts. The resulting model identifies these medical entities in other documents.\n", + "\n", + "The goal of the corpus is to advance the understanding of the causes of happiness through text-based reflection.\n", + "\n", + "Please reference [AutoML Documentation](https://cloud.google.com/natural-language/automl/docs/quickstart#model_objectives) for more information." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "KeNtSVjRxVCC" + }, + "outputs": [], + "source": [ + "# Text Extraction\n", + "IMPORT_FILE = \"gs://ucaip-test-us-central1/dataset/ucaip_ten_dataset.jsonl\"" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "4OfCqaYRqcTJ" + }, + "outputs": [], + "source": [ + "ds = aiplatform.TextDataset.create(\n", + " display_name=\"text-extraction\",\n", + " gcs_source=[IMPORT_FILE],\n", + " import_schema_uri=aiplatform.schema.dataset.ioformat.text.extraction,\n", + ")\n", + "\n", + "ds.resource_name" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "6-bBqipfqcTS" + }, + "source": [ + "## Launch a Training Job and Create a Model on Vertex AI" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aA41rT_mb-rV" + }, + "outputs": [], + "source": [ + "job = aiplatform.AutoMLTextTrainingJob(\n", + " display_name=\"text-extraction\", prediction_type=\"extraction\"\n", + ")\n", + "\n", + "# This will take around an hour to run\n", + "model = job.run(\n", + " dataset=ds,\n", + " training_fraction_split=0.6,\n", + " validation_fraction_split=0.2,\n", + " test_fraction_split=0.2,\n", + " model_display_name=\"text-extraction\",\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "5vhDsMJNqcTW" + }, + "source": [ + "# Deploy Model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Y9GH72wWqcTX" + }, + "outputs": [], + "source": [ + "endpoint = model.deploy()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "nIw1ifPuqcTb" + }, + "source": [ + "# Predict on Endpoint" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "3k6-rSZPqcTc" + }, + "outputs": [], + "source": [ + "input_text = \"\"\"\n", + "Phenotypic variation including retinitis pigmentosa, pattern dystrophy, and fundus flavimaculatus in a single family with a deletion of codon 153 or 154 of the peripherin/RDS gene.\\tBACKGROUND AND OBJECTIVES Mutations of the peripherin / RDS gene have been reported in autosomal dominant retinitis pigmentosa , pattern macular dystrophy , and retinitis punctata albescens . We report herein the occurrence of three separate phenotypes within a single family with a novel 3-base pair deletion of codon 153 or 154 of the peripherin / RDS gene . DESIGN Case reports with clinical features , fluorescein angiography , kinetic perimetry , electrophysiological studies , and molecular genetics . SETTING University medical centers . PATIENTS A 75-year-old woman , her two daughters ( aged 44 and 50 years ) , and her 49-year-old son were screened for peripherin / RDS mutations because of the presence of multiple phenotypes within the same family . RESULTS The mother presented at age 63 years with a profoundly abnormal electroretinogram ( ERG ) and adult-onset retinitis pigmentosa that progressed dramatically over 12 years , with marked loss of peripheral visual field . One daughter developed pattern macular dystrophy at age 31 years . At age 44 years , her ERG was moderately abnormal but her clinical disease was limited to the macula . Another daughter presented at age 42 years with macular degeneration and over 10 years developed the clinical picture of fundus flavimaculatus . Her peripheral visual field was preserved but her ERG was moderately abnormal . The son had onset of macular degeneration at age 44 years . Pericentral scotomas were present and the ERG was markedly abnormal . Fluorescein angiography revealed punctate pigment epithelial transmission defects . CONCLUSIONS A 3-base pair deletion of codon 153 or 154 of the peripherin / RDS gene can produce clinically disparate phenotypes even within the same family\n", + "Splicing defects in the ataxia-telangiectasia gene, ATM: underlying mutations and consequences.\\tMutations resulting in defective splicing constitute a significant proportion ( 30 / 62 [ 48 % ] ) of a new series of mutations in the ATM gene in patients with ataxia-telangiectasia ( AT ) that were detected by the protein-truncation assay followed by sequence analysis of genomic DNA . Fewer than half of the splicing mutations involved the canonical AG splice-acceptor site or GT splice-donor site . A higher percentage of mutations occurred at less stringently conserved sites , including silent mutations of the last nucleotide of exons , mutations in nucleotides other than the conserved AG and GT in the consensus splice sites , and creation of splice-acceptor or splice-donor sites in either introns or exons . These splicing mutations led to a variety of consequences , including exon skipping and , to a lesser degree , intron retention , activation of cryptic splice sites , or creation of new splice sites . In addition , 5 of 12 nonsense mutations and 1 missense mutation were associated with deletion in the cDNA of the exons in which the mutations occurred . No ATM protein was detected by western blotting in any AT cell line in which splicing mutations were identified . Several cases of exon skipping in both normal controls and patients for whom no underlying defect could be found in genomic DNA were also observed , suggesting caution in the interpretation of exon deletions observed in ATM cDNA when there is no accompanying identification of genomic mutations .\n", + "\"\"\"\n", + "\n", + "instances_list = [{\"content\": input_text}]\n", + "\n", + "prediction = endpoint.predict(instances_list)\n", + "prediction" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "RXpchK0oWqWq" + }, + "outputs": [], + "source": [ + "prediction_instance = prediction.predictions[0]\n", + "\n", + "extractions = zip(\n", + " prediction_instance[\"ids\"],\n", + " prediction_instance[\"textSegmentStartOffsets\"],\n", + " prediction_instance[\"textSegmentEndOffsets\"],\n", + " prediction_instance[\"confidences\"],\n", + " prediction_instance[\"displayNames\"],\n", + ")\n", + "\n", + "for id, start, end, confidence, display_name in extractions:\n", + " print(\n", + " f\"{id}: '{input_text[int(start):int(end)]}' predicted as '{display_name}'' with confidence {confidence}\"\n", + " )" + ] + } + ], + "metadata": { + "colab": { + "collapsed_sections": [], + "name": "AI_Platform_(Unified)_SDK_AutoML_Text_Extraction_Training.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/sdk/SDK_AutoML_Video_Action_Recognition.ipynb b/notebooks/community/sdk/SDK_AutoML_Video_Action_Recognition.ipynb new file mode 100644 index 000000000..2a5b50ff5 --- /dev/null +++ b/notebooks/community/sdk/SDK_AutoML_Video_Action_Recognition.ipynb @@ -0,0 +1,501 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "humanitarian-petite" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "corporate-remains" + }, + "source": [ + "# Feedback or issues?\n", + "\n", + "For any feedback or questions, please open an [issue](https://github.com/googleapis/python-aiplatform/issues)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "first-dietary" + }, + "source": [ + "# Vertex SDK for Python: AutoML Video Action Recognition Example\n", + "To use this Jupyter notebook, copy the notebook to a Google Cloud Notebooks instance with Tensorflow installed and open it. You can run each step, or cell, and see its results. To run a cell, use Shift+Enter. Jupyter automatically displays the return value of the last line in each cell. For more information about running notebooks in Google Cloud Notebook, see the [Google Cloud Notebook guide](https://cloud.google.com/vertex-ai/docs/general/notebooks).\n", + "\n", + "\n", + "This notebook demonstrate how to create an AutoML Video Action Recognition Model, with a Vertex AI video dataset, and how to serve the model for batch prediction. It will require you provide a bucket where the dataset will be stored.\n", + "\n", + "Note: you may incur charges for training, prediction, storage or usage of other GCP products in connection with testing this SDK." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "critical-twenty" + }, + "source": [ + "### Install Vertex SDK for Python\n", + "\n", + "\n", + "After the SDK installation the kernel will be automatically restarted." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "precious-produce" + }, + "outputs": [], + "source": [ + "!pip3 uninstall -y google-cloud-aiplatform\n", + "!pip3 install google-cloud-aiplatform\n", + "import IPython\n", + "\n", + "app = IPython.Application.instance()\n", + "app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "finished-roller" + }, + "source": [ + "### Enter Your Project and GCS Bucket\n", + "\n", + "Enter your Project Id in the cell below. Then run the cell to make sure the Cloud SDK uses the right project for all the commands in this notebook." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "limiting-costume" + }, + "outputs": [], + "source": [ + "MY_PROJECT = \"YOUR PROJECT ID\"\n", + "MY_STAGING_BUCKET = \"gs://YOUR BUCKET\" # bucket should be in same region as ucaip" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "b3GKWBB_Y6So" + }, + "outputs": [], + "source": [ + "import sys\n", + "\n", + "if \"google.colab\" in sys.modules:\n", + " import os\n", + "\n", + " from google.colab import auth\n", + "\n", + " auth.authenticate_user()\n", + " os.environ[\"GOOGLE_CLOUD_PROJECT\"] = MY_PROJECT" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "entire-fleece" + }, + "source": [ + "### Set Your Task Name, and GCS Prefix\n", + "\n", + "If you want to centeralize all input and output files under the gcs location." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "micro-administration" + }, + "outputs": [], + "source": [ + "TASK_TYPE = \"mbsdk_automl-video-training\"\n", + "PREDICTION_TYPE = \"action_recognition\"\n", + "MODEL_TYPE = \"CLOUD\"\n", + "\n", + "TASK_NAME = f\"{TASK_TYPE}_{PREDICTION_TYPE}\"\n", + "BUCKET_NAME = MY_STAGING_BUCKET.split(\"gs://\")[1]\n", + "GCS_PREFIX = TASK_NAME\n", + "\n", + "print(f\"Bucket Name: {BUCKET_NAME}\")\n", + "print(f\"Task Name: {TASK_NAME}\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "capable-sitting" + }, + "source": [ + "# HMDB: a large human motion database\n", + "We prepared some training data and prediction data for the demo using the [HMDB Dataset](https://serre-lab.clps.brown.edu/resource/hmdb-a-large-human-motion-database).\n", + "\n", + "The HMDB Dataset is licensed under the Creative Commons Attribution 4.0 International License. To view a copy of this license, visit https://creativecommons.org/licenses/by/4.0/\n", + "\n", + "For more information about this dataset please visit: https://serre-lab.clps.brown.edu/resource/hmdb-a-large-human-motion-database/" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "_NQwynxpDMrN" + }, + "outputs": [], + "source": [ + "automl_video_demo_train_data = \"gs://automl-video-demo-data/hmdb_golf_swing_all.csv\"\n", + "automl_video_demo_batch_prediction_data = (\n", + " \"gs://automl-video-demo-data/hmdb_golf_swing_predict.jsonl\"\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "catholic-financing" + }, + "source": [ + "### Copy AutoML Video Demo Train Data for Creating Managed Dataset" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "coastal-engineering" + }, + "outputs": [], + "source": [ + "gcs_source_train = f\"gs://{BUCKET_NAME}/{TASK_NAME}/data/video_action_recognition.csv\"" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "acoustic-wonder" + }, + "outputs": [], + "source": [ + "!gsutil cp $automl_video_demo_train_data $gcs_source_train" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "accepting-setup" + }, + "source": [ + "# Run AutoML Video Training with Managed Video Dataset" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "terminal-better" + }, + "source": [ + "## Initialize Vertex SDK for Python\n", + "\n", + "Initialize the *client* for Vertex AI." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "trying-mixture" + }, + "outputs": [], + "source": [ + "from google.cloud import aiplatform\n", + "\n", + "aiplatform.init(project=MY_PROJECT, staging_bucket=MY_STAGING_BUCKET)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "nonprofit-client" + }, + "source": [ + "## Create a Dataset on Vertex AI\n", + "We will now create a Vertex AI video dataset using the previously prepared csv files. Choose one of the options below. " + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "4rXbKPQHT_YS" + }, + "source": [ + "Option 1: Using MBSDK VideoDataset class" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "labeled-heating" + }, + "outputs": [], + "source": [ + "dataset = aiplatform.VideoDataset.create(\n", + " display_name=f\"temp-{TASK_NAME}\",\n", + " gcs_source=gcs_source_train,\n", + " import_schema_uri=aiplatform.schema.dataset.ioformat.video.action_recognition,\n", + " sync=False,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "_RwF8f8yURIb" + }, + "source": [ + "Option 2: Using MBSDK Dataset class\n", + "```\n", + "dataset = aiplatform.Dataset.create(\n", + " display_name=f'temp-{TASK_NAME}',\n", + " metadata_schema_uri=aiplatform.schema.dataset.metadata.video,\n", + " gcs_source=gcs_source_train, \n", + " import_schema_uri=aiplatform.schema.dataset.ioformat.video.action_recognition,\n", + " sync=False\n", + ")\n", + "```" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Kl4mKHXgVWcS" + }, + "outputs": [], + "source": [ + "dataset.wait()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "above-police" + }, + "source": [ + "## Launch a Training Job and Create a Model on Vertex AI" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "headed-saturn" + }, + "source": [ + "### Config a Training Job" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "sexual-hayes" + }, + "outputs": [], + "source": [ + "job = aiplatform.AutoMLVideoTrainingJob(\n", + " display_name=f\"temp-{TASK_NAME}\",\n", + " prediction_type=PREDICTION_TYPE,\n", + " model_type=MODEL_TYPE,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "determined-report" + }, + "source": [ + "### Run the Training Job" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "exciting-vision" + }, + "outputs": [], + "source": [ + "model = job.run(\n", + " dataset=dataset,\n", + " training_fraction_split=0.8,\n", + " test_fraction_split=0.2,\n", + " model_display_name=f\"temp-{TASK_NAME}\",\n", + " sync=False,\n", + ")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "fundamental-guess" + }, + "outputs": [], + "source": [ + "model.wait()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pregnant-charlotte" + }, + "source": [ + "# Batch Prediction Job on the Model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "outdoor-courtesy" + }, + "source": [ + "### Copy AutoML Video Demo Prediction Data for Creating Batch Prediction Job" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quarterly-sheep" + }, + "outputs": [], + "source": [ + "gcs_source_batch_prediction = f\"gs://{BUCKET_NAME}/{TASK_NAME}/data/video_action_recognition_batch_prediction.jsonl\"\n", + "gcs_destination_prefix_batch_prediction = (\n", + " f\"gs://{BUCKET_NAME}/{TASK_NAME}/batch_prediction\"\n", + ")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "declared-mexico" + }, + "outputs": [], + "source": [ + "!gsutil cp $automl_video_demo_batch_prediction_data $gcs_source_batch_prediction" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "hollywood-clearing" + }, + "outputs": [], + "source": [ + "batch_predict_job = model.batch_predict(\n", + " job_display_name=f\"temp-{TASK_NAME}\",\n", + " gcs_source=gcs_source_batch_prediction,\n", + " gcs_destination_prefix=gcs_destination_prefix_batch_prediction,\n", + " sync=False,\n", + ")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "thorough-yellow" + }, + "outputs": [], + "source": [ + "batch_predict_job.wait()\n", + "bp_iter_outputs = batch_predict_job.iter_outputs()\n", + "\n", + "prediction_results = list()\n", + "for blob in bp_iter_outputs:\n", + " if blob.name.split(\"/\")[-1].startswith(\"prediction\"):\n", + " prediction_results.append(blob.name)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "instructional-assumption" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "import tensorflow as tf\n", + "\n", + "tags = list()\n", + "for prediction_result in prediction_results:\n", + " gfile_name = f\"gs://{bp_iter_outputs.bucket.name}/{prediction_result}\"\n", + " with tf.io.gfile.GFile(name=gfile_name, mode=\"r\") as gfile:\n", + " for line in gfile.readlines():\n", + " line = json.loads(line)\n", + " break" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "processed-brooklyn" + }, + "outputs": [], + "source": [ + "line" + ] + } + ], + "metadata": { + "colab": { + "collapsed_sections": [], + "name": "AI_Platform_(Unified)_SDK_AutoML_Video_Action_Recognition.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/sdk/SDK_AutoML_Video_Classification.ipynb b/notebooks/community/sdk/SDK_AutoML_Video_Classification.ipynb new file mode 100644 index 000000000..d9197bbc3 --- /dev/null +++ b/notebooks/community/sdk/SDK_AutoML_Video_Classification.ipynb @@ -0,0 +1,496 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "modular-concentration" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "insured-graduation" + }, + "source": [ + "# Feedback or issues?\n", + "\n", + "For any feedback or questions, please open an [issue](https://github.com/googleapis/python-aiplatform/issues)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pregnant-going" + }, + "source": [ + "# Vertex SDK for Python: AutoML Video Classification Example\n", + "To use this Jupyter notebook, copy the notebook to a Google Cloud Notebooks instance with Tensorflow installed and open it. You can run each step, or cell, and see its results. To run a cell, use Shift+Enter. Jupyter automatically displays the return value of the last line in each cell. For more information about running notebooks in Google Cloud Notebook, see the [Google Cloud Notebook guide](https://cloud.google.com/vertex-ai/docs/general/notebooks).\n", + "\n", + "\n", + "This notebook demonstrate how to create an AutoML Video Classification Model, with a Vertex AI video dataset, and how to serve the model for batch prediction. It will require you provide a bucket where the dataset will be stored.\n", + "\n", + "Note: you may incur charges for training, prediction, storage or usage of other GCP products in connection with testing this SDK." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pending-chamber" + }, + "source": [ + "### Install Vertex SDK for Python\n", + "\n", + "\n", + "After the SDK installation the kernel will be automatically restarted." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "coated-remark" + }, + "outputs": [], + "source": [ + "!pip3 uninstall -y google-cloud-aiplatform\n", + "!pip3 install google-cloud-aiplatform\n", + "import IPython\n", + "\n", + "app = IPython.Application.instance()\n", + "app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "incorporated-edgar" + }, + "source": [ + "### Enter Your Project and GCS Bucket\n", + "\n", + "Enter your Project Id in the cell below. Then run the cell to make sure the Cloud SDK uses the right project for all the commands in this notebook." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "hispanic-macedonia" + }, + "outputs": [], + "source": [ + "MY_PROJECT = \"YOUR PROJECT\"\n", + "MY_STAGING_BUCKET = \"gs://YOUR BUCKET\" # bucket should be in same region as ucaip" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "efovKMU5WW7u" + }, + "outputs": [], + "source": [ + "import sys\n", + "\n", + "if \"google.colab\" in sys.modules:\n", + " import os\n", + "\n", + " from google.colab import auth\n", + "\n", + " auth.authenticate_user()\n", + " os.environ[\"GOOGLE_CLOUD_PROJECT\"] = MY_PROJECT" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "historical-consciousness" + }, + "source": [ + "### Set Your Task Name, and GCS Prefix\n", + "\n", + "If you want to centeralize all input and output files under the gcs location." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "organizational-salad" + }, + "outputs": [], + "source": [ + "TASK_TYPE = \"mbsdk_automl-video-training\"\n", + "PREDICTION_TYPE = \"classification\"\n", + "MODEL_TYPE = \"CLOUD\"\n", + "\n", + "TASK_NAME = f\"{TASK_TYPE}_{PREDICTION_TYPE}\"\n", + "BUCKET_NAME = MY_STAGING_BUCKET.split(\"gs://\")[1]\n", + "GCS_PREFIX = TASK_NAME\n", + "\n", + "print(f\"Bucket Name: {BUCKET_NAME}\")\n", + "print(f\"Task Name: {TASK_NAME}\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "compact-engagement" + }, + "source": [ + "# HMDB: a large human motion database\n", + "We prepared some training data and prediction data for the demo using the [HMDB Dataset](https://serre-lab.clps.brown.edu/resource/hmdb-a-large-human-motion-database).\n", + "\n", + "The HMDB Dataset is licensed under the Creative Commons Attribution 4.0 International License. To view a copy of this license, visit https://creativecommons.org/licenses/by/4.0/\n", + "\n", + "For more information about this dataset please visit: https://serre-lab.clps.brown.edu/resource/hmdb-a-large-human-motion-database/" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "SPDHQoFRD-vM" + }, + "outputs": [], + "source": [ + "automl_video_demo_train_data = (\n", + " \"gs://automl-video-demo-data/hmdb_split1_5classes_all.csv\"\n", + ")\n", + "automl_video_demo_batch_prediction_data = (\n", + " \"gs://automl-video-demo-data/hmdb_split1_predict.jsonl\"\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "professional-bulletin" + }, + "source": [ + "### Copy AutoML Video Demo Train Data for Creating Managed Dataset" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "accurate-producer" + }, + "outputs": [], + "source": [ + "gcs_source_train = f\"gs://{BUCKET_NAME}/{TASK_NAME}/data/video_classification.csv\"" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "sticky-casino" + }, + "outputs": [], + "source": [ + "!gsutil cp $automl_video_demo_train_data $gcs_source_train" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "rough-alert" + }, + "source": [ + "# Run AutoML Video Training with Managed Video Dataset" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "adaptive-slovakia" + }, + "source": [ + "## Initialize Vertex SDK for Python\n", + "\n", + "Initialize the *client* for Vertex AI." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "figured-fellow" + }, + "outputs": [], + "source": [ + "from google.cloud import aiplatform\n", + "\n", + "aiplatform.init(project=MY_PROJECT, staging_bucket=MY_STAGING_BUCKET)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "pleasant-holmes" + }, + "source": [ + "## Create a Dataset on Vertex AI\n", + "We will now create a Vertex AI video dataset using the previously prepared csv files. Choose one of the options below. " + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Ln-8NdHjTfbH" + }, + "source": [ + "Option 1: Using MBSDK VideoDataset class" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "uVBfL-0TTjNS" + }, + "outputs": [], + "source": [ + "dataset = aiplatform.VideoDataset.create(\n", + " display_name=f\"temp-{TASK_NAME}\",\n", + " gcs_source=gcs_source_train,\n", + " import_schema_uri=aiplatform.schema.dataset.ioformat.video.classification,\n", + " sync=False,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "lXCA_nvHTp_I" + }, + "source": [ + "Option 2: Using MBSDK Dataset class\n", + "```\n", + "dataset = aiplatform.Dataset.create(\n", + " display_name=f'temp-{TASK_NAME}',\n", + " metadata_schema_uri=aiplatform.schema.dataset.metadata.video,\n", + " gcs_source=gcs_source_train, \n", + " import_schema_uri=aiplatform.schema.dataset.ioformat.video.classification,\n", + " sync=False\n", + ")\n", + "```" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "3x4xuyIbVR_N" + }, + "outputs": [], + "source": [ + "dataset.wait()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "mexican-spending" + }, + "source": [ + "## Launch a Training Job and Create a Model on Vertex AI" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dynamic-piece" + }, + "source": [ + "### Config a Training Job" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "continuous-circular" + }, + "outputs": [], + "source": [ + "job = aiplatform.AutoMLVideoTrainingJob(\n", + " display_name=f\"temp-{TASK_NAME}\",\n", + " prediction_type=PREDICTION_TYPE,\n", + " model_type=MODEL_TYPE,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "juvenile-parameter" + }, + "source": [ + "### Run the Training Job" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "human-carrier" + }, + "outputs": [], + "source": [ + "model = job.run(\n", + " dataset=dataset,\n", + " training_fraction_split=0.8,\n", + " test_fraction_split=0.2,\n", + " model_display_name=f\"temp-{TASK_NAME}\",\n", + " sync=False,\n", + ")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "abstract-textbook" + }, + "outputs": [], + "source": [ + "model.wait()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "noted-usage" + }, + "source": [ + "# Batch Prediction Job on the Model" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ruled-smith" + }, + "source": [ + "### Copy AutoML Video Demo Prediction Data for Creating Batch Prediction Job" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "polished-dispatch" + }, + "outputs": [], + "source": [ + "gcs_source_batch_prediction = (\n", + " f\"gs://{BUCKET_NAME}/{TASK_NAME}/data/video_classification_batch_prediction.jsonl\"\n", + ")\n", + "gcs_destination_prefix_batch_prediction = (\n", + " f\"gs://{BUCKET_NAME}/{TASK_NAME}/batch_prediction\"\n", + ")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "objective-soldier" + }, + "outputs": [], + "source": [ + "!gsutil cp $automl_video_demo_batch_prediction_data $gcs_source_batch_prediction" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "piano-middle" + }, + "outputs": [], + "source": [ + "batch_predict_job = model.batch_predict(\n", + " job_display_name=f\"temp-{TASK_NAME}\",\n", + " gcs_source=gcs_source_batch_prediction,\n", + " gcs_destination_prefix=gcs_destination_prefix_batch_prediction,\n", + " sync=False,\n", + ")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "visible-scientist" + }, + "outputs": [], + "source": [ + "batch_predict_job.wait()\n", + "bp_iter_outputs = batch_predict_job.iter_outputs()\n", + "\n", + "prediction_results = list()\n", + "for blob in bp_iter_outputs:\n", + " if blob.name.split(\"/\")[-1].startswith(\"prediction\"):\n", + " prediction_results.append(blob.name)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "moving-geneva" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "import tensorflow as tf\n", + "\n", + "tags = list()\n", + "for prediction_result in prediction_results:\n", + " gfile_name = f\"gs://{bp_iter_outputs.bucket.name}/{prediction_result}\"\n", + " with tf.io.gfile.GFile(name=gfile_name, mode=\"r\") as gfile:\n", + " for line in gfile.readlines():\n", + " line = json.loads(line)\n", + " break\n", + "\n", + "print(line)" + ] + } + ], + "metadata": { + "colab": { + "collapsed_sections": [], + "name": "AI_Platform_(Unified)_SDK_AutoML_Video_Classification.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/sdk/SDK_BigQuery_Custom_Container_Training.ipynb b/notebooks/community/sdk/SDK_BigQuery_Custom_Container_Training.ipynb new file mode 100644 index 000000000..91b373778 --- /dev/null +++ b/notebooks/community/sdk/SDK_BigQuery_Custom_Container_Training.ipynb @@ -0,0 +1,521 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "a6b56b1c7b76" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "976753012196" + }, + "source": [ + "# Feedback or issues?\n", + "For any feedback or questions, please open an [issue](https://github.com/googleapis/python-aiplatform/issues)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "8c3048cd1427" + }, + "source": [ + "# Vertex SDK for Python: BigQuery Custom Container Training Example\n", + "To use this Jupyter notebook, copy the notebook to a Google Cloud Notebooks instance and open it. You can run each step, or cell, and see its results. To run a cell, use Shift+Enter. Jupyter automatically displays the return value of the last line in each cell. For more information about running notebooks in Google Cloud Notebook, see the [Google Cloud Notebook guide](https://cloud.google.com/vertex-ai/docs/general/notebooks).\n", + "\n", + "This notebook demonstrate how to create a Custom Model using Custom Container Training and a Big Query Dataset. It will require you provide a bucket where the dataset will be stored.\n", + "\n", + "Note: you may incur charges for training, prediction, storage or usage of other GCP products in connection with testing this SDK." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "JpYscdzmf4Gu" + }, + "source": [ + "# Ensure the following APIs are enabled:\n", + "- [BigQuery](https://console.cloud.google.com/apis/library/bigquery.googleapis.com?q=BigQuery)\n", + "- [Cloudbuild](https://console.cloud.google.com/apis/library/cloudbuild.googleapis.com?q=Cloudbuild)\n", + "- [Container Registry](https://console.cloud.google.com/apis/library/containerregistry.googleapis.com?q=container%20registry)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "xOMNWzTbftDr" + }, + "source": [ + "# Install Vertex SDK for Python\n", + "\n", + "\n", + "After the SDK installation the kernel will be automatically restarted." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Be020jY-ftDv" + }, + "outputs": [], + "source": [ + "!pip3 uninstall -y google-cloud-aiplatform\n", + "!pip3 install google-cloud-aiplatform\n", + "import IPython\n", + "\n", + "app = IPython.Application.instance()\n", + "app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "181b681faf5c" + }, + "source": [ + "### Enter your project and GCS bucket\n", + "\n", + "Enter your Project Id in the cell below. Then run the cell to make sure the Cloud SDK uses the right project for all the commands in this notebook." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "d61oYG3KftDw" + }, + "outputs": [], + "source": [ + "MY_PROJECT = \"YOUR PROJECT ID\"\n", + "MY_STAGING_BUCKET = \"gs://YOUR BUCKET\" # bucket should be in same region as ucaip" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "5T1d5uBoftDw" + }, + "source": [ + "# Copy Big Query Iris Dataset\n", + "We will make a Big Query dataset and copy Big Query's public iris table to that dataset. For more information about this dataset please visit: https://archive.ics.uci.edu/ml/datasets/iris " + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "DJF047yNftDw" + }, + "source": [ + "### Make BQ Dataset" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "9yOl-l_oftDx" + }, + "outputs": [], + "source": [ + "import os\n", + "\n", + "os.environ[\"GOOGLE_CLOUD_PROJECT\"] = MY_PROJECT\n", + "!bq mk {MY_PROJECT}:ml_datasets" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Xn9TuBZAftDx" + }, + "source": [ + "### Copy bigquery-public-data.ml_datasets.iris" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ISFR8nFfftDx" + }, + "outputs": [], + "source": [ + "!bq cp -n bigquery-public-data:ml_datasets.iris {MY_PROJECT}:ml_datasets.iris" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "IltwFqKIftDx" + }, + "source": [ + "# Create Training Container\n", + "We will create a directory and write all of our container build artifacts into that folder." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "40BkhtMeftDy" + }, + "outputs": [], + "source": [ + "CONTAINER_ARTIFACTS_DIR = \"demo-container-artifacts\"\n", + "\n", + "!mkdir {CONTAINER_ARTIFACTS_DIR}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "iVeG-LPOftDy" + }, + "source": [ + "### Create Cloudbuild YAML" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "4kODuFZCftDy" + }, + "outputs": [], + "source": [ + "cloudbuild_yaml = \"\"\"steps:\n", + "- name: 'gcr.io/cloud-builders/docker'\n", + " args: [ 'build', '-t', 'gcr.io/{MY_PROJECT}/test-custom-container', '.' ]\n", + "images: ['gcr.io/{MY_PROJECT}/test-custom-container']\"\"\".format(\n", + " MY_PROJECT=MY_PROJECT\n", + ")\n", + "\n", + "with open(f\"{CONTAINER_ARTIFACTS_DIR}/cloudbuild.yaml\", \"w\") as fp:\n", + " fp.write(cloudbuild_yaml)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gQ_GUCtZftDz" + }, + "source": [ + "### Write Dockerfile" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Rja_jo3rftDz" + }, + "outputs": [], + "source": [ + "%%writefile {CONTAINER_ARTIFACTS_DIR}/Dockerfile\n", + "\n", + "# Specifies base image and tag\n", + "FROM gcr.io/google-appengine/python\n", + "WORKDIR /root\n", + "\n", + "# Installs additional packages\n", + "RUN pip3 install tensorflow tensorflow-io pyarrow\n", + "\n", + "# Copies the trainer code to the docker image.\n", + "COPY test_script.py /root/test_script.py\n", + "\n", + "# Sets up the entry point to invoke the trainer.\n", + "ENTRYPOINT [\"python3\", \"test_script.py\"]" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "9dfrLShaftDz" + }, + "source": [ + "### Write entrypoint script to invoke trainer" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "S0jSd8NWftDz" + }, + "outputs": [], + "source": [ + "%%writefile {CONTAINER_ARTIFACTS_DIR}/test_script.py\n", + "\n", + "from tensorflow.python.framework import ops\n", + "from tensorflow.python.framework import dtypes\n", + "from tensorflow_io.bigquery import BigQueryClient\n", + "from tensorflow_io.bigquery import BigQueryReadSession\n", + "import tensorflow as tf\n", + "from tensorflow import feature_column\n", + "import os\n", + "\n", + "training_data_uri = os.environ[\"AIP_TRAINING_DATA_URI\"]\n", + "validation_data_uri = os.environ[\"AIP_VALIDATION_DATA_URI\"]\n", + "test_data_uri = os.environ[\"AIP_TEST_DATA_URI\"]\n", + "data_format = os.environ[\"AIP_DATA_FORMAT\"]\n", + "\n", + "def caip_uri_to_fields(uri):\n", + " uri = uri[5:]\n", + " project, dataset, table = uri.split('.')\n", + " return project, dataset, table\n", + "\n", + "feature_names = ['sepal_length', 'sepal_width', 'petal_length', 'petal_width']\n", + "\n", + "target_name = 'species'\n", + "\n", + "def transform_row(row_dict):\n", + " # Trim all string tensors\n", + " trimmed_dict = { column:\n", + " (tf.strings.strip(tensor) if tensor.dtype == 'string' else tensor) \n", + " for (column,tensor) in row_dict.items()\n", + " }\n", + " target = trimmed_dict.pop(target_name)\n", + "\n", + " target_float = tf.cond(tf.equal(tf.strings.strip(target), 'versicolor'), \n", + " lambda: tf.constant(1.0),\n", + " lambda: tf.constant(0.0))\n", + " return (trimmed_dict, target_float)\n", + "\n", + "def read_bigquery(project, dataset, table):\n", + " tensorflow_io_bigquery_client = BigQueryClient()\n", + " read_session = tensorflow_io_bigquery_client.read_session(\n", + " \"projects/\" + project,\n", + " project, table, dataset,\n", + " feature_names + [target_name],\n", + " [dtypes.float64] * 4 + [dtypes.string],\n", + " requested_streams=2)\n", + "\n", + " dataset = read_session.parallel_read_rows()\n", + " transformed_ds = dataset.map(transform_row)\n", + " return transformed_ds\n", + "\n", + "BATCH_SIZE = 16\n", + "\n", + "training_ds = read_bigquery(*caip_uri_to_fields(training_data_uri)).shuffle(10).batch(BATCH_SIZE)\n", + "eval_ds = read_bigquery(*caip_uri_to_fields(validation_data_uri)).batch(BATCH_SIZE)\n", + "test_ds = read_bigquery(*caip_uri_to_fields(test_data_uri)).batch(BATCH_SIZE)\n", + "\n", + "feature_columns = []\n", + "\n", + "# numeric cols\n", + "for header in feature_names:\n", + " feature_columns.append(feature_column.numeric_column(header))\n", + "\n", + "feature_layer = tf.keras.layers.DenseFeatures(feature_columns)\n", + "\n", + "Dense = tf.keras.layers.Dense\n", + "model = tf.keras.Sequential(\n", + " [\n", + " feature_layer,\n", + " Dense(16, activation=tf.nn.relu),\n", + " Dense(8, activation=tf.nn.relu),\n", + " Dense(4, activation=tf.nn.relu),\n", + " Dense(1, activation=tf.nn.sigmoid),\n", + " ])\n", + "\n", + "# Compile Keras model\n", + "model.compile(\n", + " loss='binary_crossentropy', \n", + " metrics=['accuracy'],\n", + " optimizer='adam')\n", + "\n", + "model.fit(training_ds, epochs=5, validation_data=eval_ds)\n", + "\n", + "print(model.evaluate(test_ds))\n", + "\n", + "tf.saved_model.save(model, os.environ[\"AIP_MODEL_DIR\"])" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "6LYlV4D2ftD0" + }, + "source": [ + "### Build Container" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "9tGdX7B_ftD1" + }, + "outputs": [], + "source": [ + "!gcloud builds submit --config {CONTAINER_ARTIFACTS_DIR}/cloudbuild.yaml {CONTAINER_ARTIFACTS_DIR}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Pf0pugbvftD1" + }, + "source": [ + "# Run Custom Container Training" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "7ee691569d8d" + }, + "source": [ + "## Initialize Vertex SDK for Python\n", + "\n", + "Initialize the *client* for Vertex AI" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "vEEr62NUftD1" + }, + "outputs": [], + "source": [ + "from google.cloud import aiplatform\n", + "\n", + "aiplatform.init(project=MY_PROJECT, staging_bucket=MY_STAGING_BUCKET)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "736ddff8408b" + }, + "source": [ + "# Create a Managed Tabular Dataset from Big Query Dataset\n", + "\n", + "This section will create a managed Tabular dataset from the iris Big Query table we copied above." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "oBdOv6lWftD1" + }, + "outputs": [], + "source": [ + "ds = aiplatform.TabularDataset.create(\n", + " display_name=\"bq_iris_dataset\", bq_source=f\"bq://{MY_PROJECT}.ml_datasets.iris\"\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ee242cc1f74c" + }, + "source": [ + "# Launch a Training Job to Create a Model\n", + "\n", + "We will train a model with the container we built above." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "uRGrFdxOftD1" + }, + "outputs": [], + "source": [ + "job = aiplatform.CustomContainerTrainingJob(\n", + " display_name=\"train-bq-iris\",\n", + " container_uri=f\"gcr.io/{MY_PROJECT}/test-custom-container:latest\",\n", + " model_serving_container_image_uri=\"gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest\",\n", + ")\n", + "model = job.run(\n", + " ds,\n", + " replica_count=1,\n", + " model_display_name=\"bq-iris-model\",\n", + " bigquery_destination=f\"bq://{MY_PROJECT}\",\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "a7fa9b59f919" + }, + "source": [ + "# Deploy Your Model\n", + "\n", + "Deploy your model, then wait until the model FINISHES deployment before proceeding to prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tEg2IDwPftD2" + }, + "outputs": [], + "source": [ + "endpoint = model.deploy(machine_type=\"n1-standard-4\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "4dbd6c650a03" + }, + "source": [ + "# Predict on the Endpoint" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "DKVhGB1PftD2" + }, + "outputs": [], + "source": [ + "endpoint.predict(\n", + " [{\"sepal_length\": 5.1, \"sepal_width\": 2.5, \"petal_length\": 3.0, \"petal_width\": 1.1}]\n", + ")" + ] + } + ], + "metadata": { + "colab": { + "collapsed_sections": [], + "name": "AI_Platform_(Unified)_SDK_BigQuery_Custom_Container_Training.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/sdk/SDK_Custom_Container_Prediction.ipynb b/notebooks/community/sdk/SDK_Custom_Container_Prediction.ipynb new file mode 100644 index 000000000..792ef3dc2 --- /dev/null +++ b/notebooks/community/sdk/SDK_Custom_Container_Prediction.ipynb @@ -0,0 +1,1118 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ur8xi4C7S06n" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "JAPoU8Sm5E6e" + }, + "source": [ + "\n", + " \n", + "
\n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tvgnzT1CKxrO" + }, + "source": [ + "## Overview\n", + "\n", + "This tutorial walks through building a custom container to serve a scikit-learn model on Vertex Predictions. You will use the FastAPI Python web server framework to create a prediction and health endpoint.\n", + "You will also cover incorporating a pre-processor from training into your online serving.\n", + "\n", + "\n", + "### Dataset\n", + "\n", + "This tutorial uses R.A. Fisher's Iris dataset, a small dataset that is popular for trying out machine learning techniques. Each instance has four numerical features, which are different measurements of a flower, and a target label that\n", + "marks it as one of three types of iris: Iris setosa, Iris versicolour, or Iris virginica.\n", + "\n", + "This tutorial uses [the copy of the Iris dataset included in the\n", + "scikit-learn library](https://scikit-learn.org/stable/datasets/index.html#iris-dataset).\n", + "\n", + "### Objective\n", + "\n", + "The goal is to:\n", + "- Train a model that uses a flower's measurements as input to predict what type of iris it is.\n", + "- Save the model and its serialized pre-processor\n", + "- Build a FastAPI server to handle predictions and health checks\n", + "- Build a custom container with model artifacts\n", + "- Upload and deploy custom container to Vertex Prediction\n", + "\n", + "This tutorial focuses more on deploying this model with Vertex AI than on\n", + "the design of the model itself.\n", + "\n", + "### Costs \n", + "\n", + "This tutorial uses billable components of Google Cloud:\n", + "\n", + "* Vertex AI\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ze4-nDLfK4pw" + }, + "source": [ + "### Set up your local development environment\n", + "\n", + "**If you are using Colab or Google Cloud Notebooks**, your environment already meets\n", + "all the requirements to run this notebook. You can skip this step." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gCuSR8GkAgzl" + }, + "source": [ + "**Otherwise**, make sure your environment meets this notebook's requirements.\n", + "You need the following:\n", + "\n", + "* Docker\n", + "* Git\n", + "* Google Cloud SDK (gcloud)\n", + "* Python 3\n", + "* virtualenv\n", + "* Jupyter notebook running in a virtual environment with Python 3\n", + "\n", + "The Google Cloud guide to [Setting up a Python development\n", + "environment](https://cloud.google.com/python/setup) and the [Jupyter\n", + "installation guide](https://jupyter.org/install) provide detailed instructions\n", + "for meeting these requirements. The following steps provide a condensed set of\n", + "instructions:\n", + "\n", + "1. [Install and initialize the Cloud SDK.](https://cloud.google.com/sdk/docs/)\n", + "\n", + "1. [Install Python 3.](https://cloud.google.com/python/setup#installing_python)\n", + "\n", + "1. [Install\n", + " virtualenv](https://cloud.google.com/python/setup#installing_and_using_virtualenv)\n", + " and create a virtual environment that uses Python 3. Activate the virtual environment.\n", + "\n", + "1. To install Jupyter, run `pip install jupyter` on the\n", + "command-line in a terminal shell.\n", + "\n", + "1. To launch Jupyter, run `jupyter notebook` on the command-line in a terminal shell.\n", + "\n", + "1. Open this notebook in the Jupyter Notebook Dashboard." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "i7EUnXsZhAGF" + }, + "source": [ + "### Install additional packages\n", + "\n", + "Install additional package dependencies not installed in your notebook environment, such as NumPy, Scikit-learn, FastAPI, Uvicorn, and joblib. Use the latest major GA version of each package." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "747f59abb3a5" + }, + "outputs": [], + "source": [ + "%%writefile requirements.txt\n", + "joblib~=1.0\n", + "numpy~=1.20\n", + "scikit-learn~=0.24\n", + "google-cloud-storage>=1.26.0,<2.0.0dev" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wyy5Lbnzg5fi" + }, + "outputs": [], + "source": [ + "# Required in Docker serving container\n", + "%pip install -U --user -r requirements.txt\n", + "\n", + "# For local FastAPI development and running\n", + "%pip install -U --user \"uvicorn[standard]>=0.12.0,<0.14.0\" fastapi~=0.63\n", + "\n", + "# Vertex SDK for Python\n", + "%pip install -U --user google-cloud-aiplatform" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "hhq5zEbGg0XX" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "After you install the additional packages, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "EzrelQZ22IZj" + }, + "outputs": [], + "source": [ + "# Automatically restart kernel after installs\n", + "import os\n", + "\n", + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + "\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "lWEdiXsJg0XY" + }, + "source": [ + "## Before you begin" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "BF1j6f9HApxa" + }, + "source": [ + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "1. [Make sure that billing is enabled for your project](https://cloud.google.com/billing/docs/how-to/modify-project).\n", + "\n", + "1. [Enable the Vertex AI API and Compute Engine API](https://console.cloud.google.com/flows/enableapi?apiid=aiplatform.googleapis.com,compute_component).\n", + "\n", + "1. If you are running this notebook locally, you will need to install the [Cloud SDK](https://cloud.google.com/sdk).\n", + "\n", + "1. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` or `%` as shell commands, and it interpolates Python variables with `$` or `{}` into these commands." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "WReHDGG5g0XY" + }, + "source": [ + "#### Set your project ID\n", + "\n", + "**If you don't know your project ID**, you may be able to get your project ID using `gcloud`." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "oM1iC_MfAts1" + }, + "outputs": [], + "source": [ + "# Get your Google Cloud project ID from gcloud\n", + "shell_output=!gcloud config list --format 'value(core.project)' 2>/dev/null\n", + "\n", + "try:\n", + " PROJECT_ID = shell_output[0]\n", + "except IndexError:\n", + " PROJECT_ID = None\n", + "\n", + "print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "qJYoRfYng0XZ" + }, + "source": [ + "Otherwise, set your project ID here." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "riG_qUokg0XZ" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None:\n", + " PROJECT_ID = \"[your-project-id]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dr--iN2kAylZ" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebooks**, your environment is already\n", + "authenticated. Skip this step." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sBCra4QMA2wR" + }, + "source": [ + "**If you are using Colab**, run the cell below and follow the instructions\n", + "when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "1. In the Cloud Console, go to the [**Create service account key**\n", + " page](https://console.cloud.google.com/apis/credentials/serviceaccountkey).\n", + "\n", + "2. Click **Create service account**.\n", + "\n", + "3. In the **Service account name** field, enter a name, and\n", + " click **Create**.\n", + "\n", + "4. In the **Grant this service account access to project** section, click the **Role** drop-down list. Type \"Vertex AI\"\n", + "into the filter box, and select\n", + " **Vertex AI Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "5. Click *Create*. A JSON file that contains your key downloads to your\n", + "local environment.\n", + "\n", + "6. Enter the path to your service account key as the\n", + "`GOOGLE_APPLICATION_CREDENTIALS` variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "PyQmSRbKA8r-" + }, + "outputs": [], + "source": [ + "import os\n", + "import sys\n", + "\n", + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebooks, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\") and not os.getenv(\n", + " \"GOOGLE_APPLICATION_CREDENTIALS\"\n", + " ):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "XoEqT2Y4DJmf" + }, + "source": [ + "### Configure project and resource names" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "MzGDU7TWdts_" + }, + "outputs": [], + "source": [ + "REGION = \"us-central1\" # @param {type:\"string\"}\n", + "MODEL_ARTIFACT_DIR = \"custom-container-prediction-model\" # @param {type:\"string\"}\n", + "REPOSITORY = \"custom-container-prediction\" # @param {type:\"string\"}\n", + "IMAGE = \"sklearn-fastapi-server\" # @param {type:\"string\"}\n", + "MODEL_DISPLAY_NAME = \"sklearn-custom-container\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ca1a915d641d" + }, + "source": [ + "`REGION` - Used for operations\n", + "throughout the rest of this notebook. Make sure to [choose a region where Cloud\n", + "Vertex AI services are\n", + "available](https://cloud.google.com/vertex-ai/docs/general/locations#feature-availability). You may\n", + "not use a Multi-Regional Storage bucket for training with Vertex AI.\n", + "\n", + "`MODEL_ARTIFACT_DIR` - Folder directory path to your model artifacts within a Cloud Storage bucket, for example: \"my-models/fraud-detection/trial-4\"\n", + "\n", + "`REPOSITORY` - Name of the Artifact Repository to create or use.\n", + "\n", + "`IMAGE` - Name of the container image that will be pushed.\n", + "\n", + "`MODEL_DISPLAY_NAME` - Display name of Vertex AI Model resource." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "62f861b68b50" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "To update your model artifacts without re-building the container, you must upload your model\n", + "artifacts and any custom code to Cloud Storage.\n", + "\n", + "Set the name of your Cloud Storage bucket below. It must be unique across all\n", + "Cloud Storage buckets. " + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "9724b00aeead" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" # @param {type:\"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "58cb4f5895f0" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "2d2208676cee" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "c664a5abc11a" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "2c1b1c29f5f6" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "3c2d091d9e73" + }, + "source": [ + "## Write your pre-processor\n", + "Scaling training data so each numerical feature column has a mean of 0 and a standard deviation of 1 [can improve your model](https://developers.google.com/machine-learning/crash-course/representation/cleaning-data).\n", + "\n", + "Create `preprocess.py`, which contains a class to do this scaling:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "6e74556ea0b4" + }, + "outputs": [], + "source": [ + "%mkdir app" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "58d843d21fa8" + }, + "outputs": [], + "source": [ + "%%writefile app/preprocess.py\n", + "import numpy as np\n", + "\n", + "class MySimpleScaler(object):\n", + " def __init__(self):\n", + " self._means = None\n", + " self._stds = None\n", + "\n", + " def preprocess(self, data):\n", + " if self._means is None: # during training only\n", + " self._means = np.mean(data, axis=0)\n", + "\n", + " if self._stds is None: # during training only\n", + " self._stds = np.std(data, axis=0)\n", + " if not self._stds.all():\n", + " raise ValueError(\"At least one column has standard deviation of 0.\")\n", + "\n", + " return (data - self._means) / self._stds\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "4b816cd52f4b" + }, + "source": [ + "## Train and store model with pre-processor\n", + "Next, use `preprocess.MySimpleScaler` to preprocess the iris data, then train a model using scikit-learn.\n", + "\n", + "At the end, export your trained model as a joblib (`.joblib`) file and export your `MySimpleScaler` instance as a pickle (`.pkl`) file:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "43e47249f736" + }, + "outputs": [], + "source": [ + "%cd app/\n", + "\n", + "import pickle\n", + "\n", + "import joblib\n", + "from preprocess import MySimpleScaler\n", + "from sklearn.datasets import load_iris\n", + "from sklearn.ensemble import RandomForestClassifier\n", + "\n", + "iris = load_iris()\n", + "scaler = MySimpleScaler()\n", + "\n", + "X = scaler.preprocess(iris.data)\n", + "y = iris.target\n", + "\n", + "model = RandomForestClassifier()\n", + "model.fit(X, y)\n", + "\n", + "joblib.dump(model, \"model.joblib\")\n", + "with open(\"preprocessor.pkl\", \"wb\") as f:\n", + " pickle.dump(scaler, f)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "3849066a33bd" + }, + "source": [ + "### Upload model artifacts and custom code to Cloud Storage\n", + "\n", + "Before you can deploy your model for serving, Vertex AI needs access to the following files in Cloud Storage:\n", + "\n", + "* `model.joblib` (model artifact)\n", + "* `preprocessor.pkl` (model artifact)\n", + "\n", + "Run the following commands to upload your files:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ca67ee52d4d9" + }, + "outputs": [], + "source": [ + "!gsutil cp model.joblib preprocessor.pkl {BUCKET_NAME}/{MODEL_ARTIFACT_DIR}/\n", + "%cd .." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "480a1d88ecdb" + }, + "source": [ + "## Build a FastAPI server" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "94af0ba5eadd" + }, + "outputs": [], + "source": [ + "%%writefile app/main.py\n", + "from fastapi import FastAPI, Request\n", + "\n", + "import joblib\n", + "import json\n", + "import numpy as np\n", + "import pickle\n", + "import os\n", + "\n", + "from google.cloud import storage\n", + "from preprocess import MySimpleScaler\n", + "from sklearn.datasets import load_iris\n", + "\n", + "\n", + "app = FastAPI()\n", + "gcs_client = storage.Client()\n", + "\n", + "with open(\"preprocessor.pkl\", 'wb') as preprocessor_f, open(\"model.joblib\", 'wb') as model_f:\n", + " gcs_client.download_blob_to_file(\n", + " f\"{os.environ['AIP_STORAGE_URI']}/preprocessor.pkl\", preprocessor_f\n", + " )\n", + " gcs_client.download_blob_to_file(\n", + " f\"{os.environ['AIP_STORAGE_URI']}/model.joblib\", model_f\n", + " )\n", + "\n", + "with open(\"preprocessor.pkl\", \"rb\") as f:\n", + " preprocessor = pickle.load(f)\n", + "\n", + "_class_names = load_iris().target_names\n", + "_model = joblib.load(\"model.joblib\")\n", + "_preprocessor = preprocessor\n", + "\n", + "\n", + "@app.get(os.environ['AIP_HEALTH_ROUTE'], status_code=200)\n", + "def health():\n", + " return {}\n", + "\n", + "\n", + "@app.post(os.environ['AIP_PREDICT_ROUTE'])\n", + "async def predict(request: Request):\n", + " body = await request.json()\n", + "\n", + " instances = body[\"instances\"]\n", + " inputs = np.asarray(instances)\n", + " preprocessed_inputs = _preprocessor.preprocess(inputs)\n", + " outputs = _model.predict(preprocessed_inputs)\n", + "\n", + " return {\"predictions\": [_class_names[class_num] for class_num in outputs]}\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "469f55daf250" + }, + "source": [ + "### Add pre-start script\n", + "FastAPI will execute this script before starting up the server. The `PORT` environment variable is set to equal `AIP_HTTP_PORT` in order to run FastAPI on same the port expected by Vertex AI." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "69f438aca35b" + }, + "outputs": [], + "source": [ + "%%writefile app/prestart.sh\n", + "#!/bin/bash\n", + "export PORT=$AIP_HTTP_PORT" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "8b62ddf1def3" + }, + "source": [ + "### Store test instances to use later\n", + "To learn more about formatting input instances in JSON, [read the documentation.](https://cloud.google.com/vertex-ai/docs/predictions/online-predictions-custom-models#request-body-details)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "b6605e9e6186" + }, + "outputs": [], + "source": [ + "%%writefile instances.json\n", + "{\n", + " \"instances\": [\n", + " [6.7, 3.1, 4.7, 1.5],\n", + " [4.6, 3.1, 1.5, 0.2]\n", + " ]\n", + "}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "51e149fdec1b" + }, + "source": [ + "## Build and push container to Artifact Registry" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "3bdb9a7768a5" + }, + "source": [ + "### Build your container\n", + "Optionally copy in your credentials to run the container locally." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "fbb77f4f56c7" + }, + "outputs": [], + "source": [ + "# NOTE: Copy in credentials to run locally, this step can be skipped for deployment\n", + "%cp $GOOGLE_APPLICATION_CREDENTIALS app/credentials.json" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "240578ec9efe" + }, + "source": [ + "Write the Dockerfile, using `tiangolo/uvicorn-gunicorn-fastapi` as a base image. This will automatically run FastAPI for you using Gunicorn and Uvicorn. Visit [the FastAPI docs to read more about deploying FastAPI with Docker](https://fastapi.tiangolo.com/deployment/docker/)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "3d3a6b9ed22b" + }, + "outputs": [], + "source": [ + "%%writefile Dockerfile\n", + "\n", + "FROM tiangolo/uvicorn-gunicorn-fastapi:python3.7\n", + "\n", + "COPY ./app /app\n", + "COPY requirements.txt requirements.txt\n", + "\n", + "RUN pip install -r requirements.txt" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "04c988201499" + }, + "source": [ + "Build the image and tag the Artifact Registry path that you will push to." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "f1e7d639b9cc" + }, + "outputs": [], + "source": [ + "!docker build \\\n", + " --tag={REGION}-docker.pkg.dev/{PROJECT_ID}/{REPOSITORY}/{IMAGE} \\\n", + " ." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "147a555f6c93" + }, + "source": [ + "### Run and test the container locally (optional)\n", + "\n", + "Run the container locally in detached mode and provide the environment variables that the container requires. These env vars will be provided to the container by Vertex Prediction once deployed. Test the `/health` and `/predict` routes, then stop the running image." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "62ed2d334d0f" + }, + "outputs": [], + "source": [ + "!docker rm local-iris\n", + "!docker run -d -p 80:8080 \\\n", + " --name=local-iris \\\n", + " -e AIP_HTTP_PORT=8080 \\\n", + " -e AIP_HEALTH_ROUTE=/health \\\n", + " -e AIP_PREDICT_ROUTE=/predict \\\n", + " -e AIP_STORAGE_URI={BUCKET_NAME}/{MODEL_ARTIFACT_DIR} \\\n", + " -e GOOGLE_APPLICATION_CREDENTIALS=credentials.json \\\n", + " {REGION}-docker.pkg.dev/{PROJECT_ID}/{REPOSITORY}/{IMAGE}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ce629eea32fd" + }, + "outputs": [], + "source": [ + "!curl localhost/health" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "56986f93438e" + }, + "outputs": [], + "source": [ + "!curl -X POST \\\n", + " -d @instances.json \\\n", + " -H \"Content-Type: application/json; charset=utf-8\" \\\n", + " localhost/predict" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "a29fcbbe0188" + }, + "outputs": [], + "source": [ + "!docker stop local-iris" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "212b2935ea12" + }, + "source": [ + "### Push the container to artifact registry\n", + "\n", + "Configure Docker to access Artifact Registry. Then push your container image to your Artifact Registry repository." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "09ffe2434e3d" + }, + "outputs": [], + "source": [ + "!gcloud beta artifacts repositories create {REPOSITORY} \\\n", + " --repository-format=docker \\\n", + " --location=$REGION" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "293437024749" + }, + "outputs": [], + "source": [ + "!gcloud auth configure-docker {REGION}-docker.pkg.dev" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "1dd7448f4703" + }, + "outputs": [], + "source": [ + "!docker push {REGION}-docker.pkg.dev/{PROJECT_ID}/{REPOSITORY}/{IMAGE}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "b438bfa2129f" + }, + "source": [ + "## Deploy to Vertex AI\n", + "\n", + "Use the Python SDK to upload and deploy your model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "4ae19df6a33e" + }, + "source": [ + "### Upload the custom container model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "8d682d8388ec" + }, + "outputs": [], + "source": [ + "from google.cloud import aiplatform" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "574fb82d3eed" + }, + "outputs": [], + "source": [ + "aiplatform.init(project=PROJECT, location=REGION)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "2738154345d5" + }, + "outputs": [], + "source": [ + "model = aiplatform.Model.upload(\n", + " display_name=MODEL_DISPLAY_NAME,\n", + " artifact_uri=f\"{BUCKET_NAME}/{MODEL_ARTIFACT_DIR}\",\n", + " serving_container_image_uri=f\"{REGION}-docker.pkg.dev/{PROJECT_ID}/{REPOSITORY}/{IMAGE}\",\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bd1b85afc7df" + }, + "source": [ + "### Deploy the model on Vertex AI\n", + "After this step completes, the model is deployed and ready for online prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "62cf66498a28" + }, + "outputs": [], + "source": [ + "endpoint = model.deploy(machine_type=\"n1-standard-4\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "6883e7b07143" + }, + "source": [ + "## Send predictions\n", + "\n", + "### Using Python SDK" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "d69ed411c2d3" + }, + "outputs": [], + "source": [ + "endpoint.predict(instances=[[6.7, 3.1, 4.7, 1.5], [4.6, 3.1, 1.5, 0.2]])" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "370d22f53427" + }, + "source": [ + "### Using REST" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ba55bc560d58" + }, + "outputs": [], + "source": [ + "ENDPOINT_ID = endpoint.name" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "95c562b4e98b" + }, + "outputs": [], + "source": [ + "! curl \\\n", + "-H \"Authorization: Bearer $(gcloud auth print-access-token)\" \\\n", + "-H \"Content-Type: application/json\" \\\n", + "-d @instances.json \\\n", + "https://{REGION}-aiplatform.googleapis.com/v1/projects/{PROJECT_ID}/locations/{REGION}/endpoints/{ENDPOINT_ID}:predict" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "fa71174a7dd0" + }, + "source": [ + "### Using gcloud CLI" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "23b8e807b02c" + }, + "outputs": [], + "source": [ + "!gcloud beta ai endpoints predict $ENDPOINT_ID \\\n", + " --region=$REGION \\\n", + " --json-request=instances.json" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "TpV-iwP9qw9c" + }, + "source": [ + "## Cleaning up\n", + "\n", + "To clean up all Google Cloud resources used in this project, you can [delete the Google Cloud\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "sx_vKniMq9ZX" + }, + "outputs": [], + "source": [ + "# Undeploy model and delete endpoint\n", + "endpoint.delete(force=True)\n", + "\n", + "# Delete the model resource\n", + "model.delete()\n", + "\n", + "# Delete the container image from Artifact Registry\n", + "!gcloud artifacts docker images delete \\\n", + " --quiet \\\n", + " --delete-tags \\\n", + " {REGION}-docker.pkg.dev/{PROJECT_ID}/{REPOSITORY}/{IMAGE}" + ] + } + ], + "metadata": { + "colab": { + "collapsed_sections": [], + "name": "AI_Platform_(Unified)_SDK_Custom_Container_Prediction.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/sdk/SDK_Custom_Model_Training_AutoML_Tabular_Model_Training.ipynb b/notebooks/community/sdk/SDK_Custom_Model_Training_AutoML_Tabular_Model_Training.ipynb new file mode 100644 index 000000000..4d28d2339 --- /dev/null +++ b/notebooks/community/sdk/SDK_Custom_Model_Training_AutoML_Tabular_Model_Training.ipynb @@ -0,0 +1,452 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ur8xi4C7S06n" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "1c955a52adac" + }, + "source": [ + "# Feedback or issues?\n", + "For any feedback or questions, please open an [issue](https://github.com/googleapis/python-aiplatform/issues)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "eHLV0D7Y5jtU" + }, + "source": [ + "# Vertex SDK for Python: Custom Training and AutoML Tabular Training Example\n", + "\n", + "To use this Colaboratory notebook, you copy the notebook to your own Google Drive and open it with Colaboratory (or Colab). You can run each step, or cell, and see its results. To run a cell, use Shift+Enter. Colab automatically displays the return value of the last line in each cell. For more information about running notebooks in Colab, see the [Colab welcome page](https://colab.research.google.com/notebooks/welcome.ipynb).\n", + "\n", + "This notebook demonstrate how to create both an AutoML model and a custom model based on a tabular dataset. It will require you provide a bucket where the dataset will be stored.\n", + "\n", + "Note: you may incur charges for training, prediction, storage or usage of other GCP products in connection with testing this SDK." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "lld3eeJUs5yM" + }, + "source": [ + "# Install Vertex SDK for Python, Authenticate, and upload of a Dataset to your GCS bucket\n", + "\n", + "\n", + "After the SDK installation the kernel will be automatically restarted. You may see this error message `Your session crashed for an unknown reason` which is normal." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "sBfZtR4X1Dr_" + }, + "outputs": [], + "source": [ + "!pip3 uninstall -y google-cloud-aiplatform\n", + "!pip3 install google-cloud-aiplatform\n", + "import IPython\n", + "\n", + "app = IPython.Application.instance()\n", + "app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "0117fe65129c" + }, + "outputs": [], + "source": [ + "import sys\n", + "\n", + "if \"google.colab\" in sys.modules:\n", + " import os\n", + "\n", + " from google.colab import auth\n", + "\n", + " auth.authenticate_user()\n", + " os.environ[\"GOOGLE_CLOUD_PROJECT\"] = MY_PROJECT" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "c0SNmTBeD2nV" + }, + "source": [ + "### Enter your project and GCS bucket\n", + "\n", + "Enter your Project Id in the cell below. Then run the cell to make sure the Cloud SDK uses the right project for all the commands in this notebook." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "iqSQT6Z6bekX" + }, + "outputs": [], + "source": [ + "MY_PROJECT = \"YOUR PROJECT\"\n", + "MY_STAGING_BUCKET = \"gs://YOUR BUCKET\" # bucket should be in same region as Vertex AI" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "zaQRsZ7vwkmE" + }, + "source": [ + "The dataset we are using is the California Housing Dataset, available locally in Colab. For more information about this dataset and its license, please visit: https://developers.google.com/machine-learning/crash-course/california-housing-data-description" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "V_T10yTTqcS_" + }, + "outputs": [], + "source": [ + "!gcloud config set project {MY_PROJECT}\n", + "\n", + "TRAIN_FILE_NAME = \"california_housing_train.csv\"\n", + "!gsutil cp sample_data/{TRAIN_FILE_NAME} {MY_STAGING_BUCKET}/data/\n", + "\n", + "gcs_csv_path = f\"{MY_STAGING_BUCKET}/data/{TRAIN_FILE_NAME}\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "rk43VP_IqcTE" + }, + "source": [ + "# Initialize Vertex SDK for Python\n", + "\n", + "Initialize the *client* for Vertex AI" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "VCiC9gBWqcTF" + }, + "outputs": [], + "source": [ + "from google.cloud import aiplatform\n", + "\n", + "aiplatform.init(project=MY_PROJECT, staging_bucket=MY_STAGING_BUCKET)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "35QVNhACqcTJ" + }, + "source": [ + "# Create Managed Tabular Dataset from CSV" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "4OfCqaYRqcTJ" + }, + "outputs": [], + "source": [ + "ds = aiplatform.TabularDataset.create(\n", + " display_name=\"housing\", gcs_source=[gcs_csv_path], sync=False\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Iw7wMxqNOBL6" + }, + "source": [ + "# Write your Training Script\n", + "- Write this cell as a file which will be used for custom training." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "hXJDoIkaOTKu" + }, + "outputs": [], + "source": [ + "%%writefile training_script.py\n", + "\n", + "import pandas as pd\n", + "import os\n", + "import tensorflow as tf\n", + "from tensorflow import keras\n", + "from tensorflow.keras import layers\n", + "\n", + "\n", + "# uncomment and bump up replica_count for distributed training\n", + "# strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "# tf.distribute.experimental_set_strategy(strategy)\n", + "\n", + "target = 'median_house_value'\n", + "\n", + "def aip_data_to_dataframe(wild_card_path):\n", + " return pd.concat([pd.read_csv(fp.numpy().decode())\n", + " for fp in tf.data.Dataset.list_files([wild_card_path])])\n", + "\n", + "def get_features_and_labels(df):\n", + " features = df.drop(target, axis=1) \n", + " return {key: features[key].values for key in features.columns}, df[target].values\n", + "\n", + "def data_prep(wild_card_path):\n", + " return get_features_and_labels(aip_data_to_dataframe(wild_card_path))\n", + "\n", + "train_features, train_labels = data_prep(os.environ[\"AIP_TRAINING_DATA_URI\"])\n", + "\n", + "feature_columns = [tf.feature_column.numeric_column(name) for name in \n", + " train_features.keys()]\n", + "\n", + "model = tf.keras.Sequential([\n", + " layers.DenseFeatures(feature_columns),\n", + " layers.Dense(64),\n", + " layers.Dense(1)])\n", + "model.compile(loss='mse', optimizer='adam')\n", + "\n", + "model.fit(train_features, train_labels,\n", + " epochs=10,\n", + " validation_data=data_prep(os.environ[\"AIP_VALIDATION_DATA_URI\"]))\n", + "print(model.evaluate(*data_prep(os.environ[\"AIP_TEST_DATA_URI\"])))\n", + "\n", + "# save as Vertex AI Managed model\n", + "tf.saved_model.save(model, os.environ[\"AIP_MODEL_DIR\"])" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "6-bBqipfqcTS" + }, + "source": [ + "# Launch Training AutoML and Custom Training Jobs to Create Models\n", + "\n", + "Once we have created your dataset and defined your training script, we will create a model." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "aA41rT_mb-rV" + }, + "outputs": [], + "source": [ + "custom_job = aiplatform.CustomTrainingJob(\n", + " display_name=\"train-housing\",\n", + " script_path=\"training_script.py\",\n", + " container_uri=\"gcr.io/cloud-aiplatform/training/tf-cpu.2-2:latest\",\n", + " requirements=[\n", + " \"gcsfs==0.7.1\",\n", + " ],\n", + " model_serving_container_image_uri=\"gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest\",\n", + ")\n", + "\n", + "custom_model = custom_job.run(\n", + " ds, replica_count=1, model_display_name=\"housing-model\", sync=False\n", + ")\n", + "\n", + "automl_job = aiplatform.AutoMLTabularTrainingJob(\n", + " display_name=\"train-housing-automl_1\",\n", + " optimization_prediction_type=\"regression\",\n", + " optimization_objective=\"minimize-rmse\",\n", + " column_transformations=[\n", + " {\"numeric\": {\"column_name\": \"longitude\"}},\n", + " {\"numeric\": {\"column_name\": \"latitude\"}},\n", + " {\"numeric\": {\"column_name\": \"housing_median_age\"}},\n", + " {\"numeric\": {\"column_name\": \"total_rooms\"}},\n", + " {\"numeric\": {\"column_name\": \"total_bedrooms\"}},\n", + " {\"numeric\": {\"column_name\": \"population\"}},\n", + " {\"numeric\": {\"column_name\": \"households\"}},\n", + " {\"numeric\": {\"column_name\": \"median_income\"}},\n", + " ],\n", + " optimization_objective_recall_value=None,\n", + " optimization_objective_precision_value=None,\n", + ")\n", + "\n", + "# This will take around an hour to run\n", + "automl_model = automl_job.run(\n", + " dataset=ds,\n", + " target_column=\"median_house_value\",\n", + " training_fraction_split=0.6,\n", + " validation_fraction_split=0.2,\n", + " test_fraction_split=0.2,\n", + " weight_column=None,\n", + " budget_milli_node_hours=1000,\n", + " model_display_name=\"house-value-prediction-model\",\n", + " disable_early_stopping=False,\n", + " predefined_split_column_name=None,\n", + " sync=False,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "5vhDsMJNqcTW" + }, + "source": [ + "# Deploy Model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Y9GH72wWqcTX" + }, + "outputs": [], + "source": [ + "custom_endpoint = custom_model.deploy(machine_type=\"n1-standard-4\", sync=False)\n", + "automl_endpoint = automl_model.deploy(machine_type=\"n1-standard-4\", sync=False)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "nIw1ifPuqcTb" + }, + "source": [ + "# Predict on Endpoint" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "KDd6VKWDIRgh" + }, + "outputs": [], + "source": [ + "custom_endpoint.wait()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "5MssDOgyIPIK" + }, + "outputs": [], + "source": [ + "# This sample is taken from an observation where median_house_value = 94600\n", + "prediction = custom_endpoint.predict(\n", + " [\n", + " {\n", + " \"longitude\": -124.35,\n", + " \"latitude\": 40.54,\n", + " \"housing_median_age\": 52.0,\n", + " \"total_rooms\": 1820.0,\n", + " \"total_bedrooms\": 300.0,\n", + " \"population\": 806,\n", + " \"households\": 270.0,\n", + " \"median_income\": 3.014700,\n", + " },\n", + " ]\n", + ")\n", + "prediction" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "vNFA7MmFf_zd" + }, + "outputs": [], + "source": [ + "automl_job.state" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "KjVy8tFXgC6w" + }, + "outputs": [], + "source": [ + "automl_endpoint.wait()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "3k6-rSZPqcTc" + }, + "outputs": [], + "source": [ + "# This sample is taken from an observation where median_house_value = 94600\n", + "prediction = automl_endpoint.predict(\n", + " [\n", + " {\n", + " \"longitude\": \"-124.35\",\n", + " \"latitude\": \"40.54\",\n", + " \"housing_median_age\": \"52.0\",\n", + " \"total_rooms\": \"1820.0\",\n", + " \"total_bedrooms\": \"300.0\",\n", + " \"population\": \"806\",\n", + " \"households\": \"270.0\",\n", + " \"median_income\": \"3.014700\",\n", + " },\n", + " ]\n", + ")\n", + "prediction" + ] + } + ], + "metadata": { + "colab": { + "collapsed_sections": [], + "name": "AI_Platform_(Unified)_SDK_Custom_Model_Training_AutoML_Tabular_Model_Training.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/sdk/SDK_Custom_Training_Python_Package_Managed_Text_Dataset_Tensorflow_Serving_Container.ipynb b/notebooks/community/sdk/SDK_Custom_Training_Python_Package_Managed_Text_Dataset_Tensorflow_Serving_Container.ipynb new file mode 100644 index 000000000..5634851c4 --- /dev/null +++ b/notebooks/community/sdk/SDK_Custom_Training_Python_Package_Managed_Text_Dataset_Tensorflow_Serving_Container.ipynb @@ -0,0 +1,1259 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "41aMjXA5rKoL" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bcKMGrqDrKoS" + }, + "source": [ + "# Feedback or issues?\n", + "\n", + "For any feedback or questions, please open an [issue](https://github.com/googleapis/python-aiplatform/issues)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ij53c3EzrKoS" + }, + "source": [ + "# Vertex SDK for Python: Custom Training using Python Package, Managed Text Dataset, and TF-Serving Container Example\n", + "To use this Jupyter notebook, copy the notebook to a Google Cloud Notebooks instance with Tensorflow installed and open it. You can run each step, or cell, and see its results. To run a cell, use Shift+Enter. Jupyter automatically displays the return value of the last line in each cell. For more information about running notebooks in Google Cloud Notebook, see the [Google Cloud Notebook guide](https://cloud.google.com/vertex-ai/docs/general/notebooks).\n", + "\n", + "This notebook demonstrate how to create a Custom Model using Custom Python Package Training, with a Vertex AI Dataset, and how to serve the model using Tensorflow-Serving Container for online prediction, and batch prediction. It will require you provide a bucket where the dataset will be stored.\n", + "\n", + "Note: you may incur charges for training, prediction, storage or usage of other GCP products in connection with testing this SDK." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "xOMNWzTbftDr" + }, + "source": [ + "### Install Vertex SDK for Python\n", + "\n", + "\n", + "After the SDK installation the kernel will be automatically restarted." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Be020jY-ftDv" + }, + "outputs": [], + "source": [ + "!pip3 uninstall -y google-cloud-aiplatform\n", + "!pip3 install google-cloud-aiplatform\n", + "import IPython\n", + "\n", + "app = IPython.Application.instance()\n", + "app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "l6N5ru3xrKoU" + }, + "source": [ + "### Enter Your Project and GCS Bucket\n", + "\n", + "Enter your Project Id in the cell below. Then run the cell to make sure the Cloud SDK uses the right project for all the commands in this notebook." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "d61oYG3KftDw" + }, + "outputs": [], + "source": [ + "MY_PROJECT = \"YOUR PROJECT ID\"\n", + "MY_STAGING_BUCKET = \"gs://YOUR BUCKET\" # bucket should be in same region as ucaip" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "YHNj4P3vrKoV" + }, + "source": [ + "### Set Your Application Name, Task Name, and Directories.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "RnNL5St8rKoV" + }, + "outputs": [], + "source": [ + "APP_NAME = \"keras-text-class-stack-overflow-tag\"\n", + "TASK_TYPE = \"mbsdk_custom-py-pkg-training\"\n", + "\n", + "TASK_NAME = f\"{TASK_TYPE}_{APP_NAME}\"\n", + "\n", + "TASK_DIR = f\"./{TASK_NAME}\"\n", + "DATA_DIR = f\"{TASK_DIR}/data\"\n", + "\n", + "print(f\"Task Name: {TASK_NAME}\")\n", + "print(f\"Task Directory: {TASK_DIR}\")\n", + "print(f\"Data Directory: {DATA_DIR}\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "9hZmDcfLS13W" + }, + "source": [ + "### Set a GCS Prefix\n", + "\n", + "If you want to centeralize all input and output files under the gcs location." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "W_4gzaP0SyQd" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = MY_STAGING_BUCKET.split(\"gs://\")[1]\n", + "GCS_PREFIX = f\"{TASK_TYPE}/{APP_NAME}\"\n", + "\n", + "print(f\"Bucket Name: {BUCKET_NAME}\")\n", + "print(f\"GCS Prefix: {GCS_PREFIX}\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "5T1d5uBoftDw" + }, + "source": [ + "# Stack Overflow Data\n", + "We download the stack overflow data from from https://storage.googleapis.com/download.tensorflow.org/data/stack_overflow_16k.tar.gz and will create a Vertex AI managed text dataset. \n", + "\n", + "The Stack Overflow Data is licensed under the Creative Commons Attribution-ShareAlike 3.0 Unported License. To view a copy of this license, visit http://creativecommons.org/licenses/by-sa/3.0/ \n", + "\n", + "For more information about this dataset please visit: https://console.cloud.google.com/marketplace/details/stack-exchange/stack-overflow\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "DJF047yNftDw" + }, + "source": [ + "### Utility Functions to Download Data and Prepare CSV Files for Creating Vertex AI Managed Dataset" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "9yOl-l_oftDx" + }, + "outputs": [], + "source": [ + "import csv\n", + "import os\n", + "\n", + "from google.cloud import storage\n", + "from tensorflow.keras import utils\n", + "\n", + "\n", + "def upload_blob(bucket_name, source_file_name, destination_blob_name):\n", + " \"\"\"Uploads a file to the bucket.\"\"\"\n", + "\n", + " storage_client = storage.Client()\n", + " bucket = storage_client.bucket(bucket_name)\n", + " blob = bucket.blob(destination_blob_name)\n", + "\n", + " blob.upload_from_filename(source_file_name)\n", + "\n", + " destination_file_name = os.path.join(\"gs://\", bucket_name, destination_blob_name)\n", + "\n", + " return destination_file_name\n", + "\n", + "\n", + "def download_data(data_dir):\n", + " \"\"\"Download data.\"\"\"\n", + "\n", + " if not os.path.exists(data_dir):\n", + " os.makedirs(data_dir)\n", + "\n", + " url = \"https://storage.googleapis.com/download.tensorflow.org/data/stack_overflow_16k.tar.gz\"\n", + " dataset = utils.get_file(\n", + " \"stack_overflow_16k.tar.gz\",\n", + " url,\n", + " untar=True,\n", + " cache_dir=data_dir,\n", + " cache_subdir=\"\",\n", + " )\n", + " data_dir = os.path.join(os.path.dirname(dataset))\n", + "\n", + " return data_dir\n", + "\n", + "\n", + "def upload_train_data_to_gcs(train_data_dir, bucket_name, destination_blob_prefix):\n", + " \"\"\"Create CSV file using train data content.\"\"\"\n", + "\n", + " train_data_dir = os.path.join(data_dir, \"train\")\n", + " train_data_fn = os.path.join(data_dir, \"train.csv\")\n", + "\n", + " fp = open(train_data_fn, \"w\", encoding=\"utf8\")\n", + " writer = csv.writer(\n", + " fp, delimiter=\",\", quotechar='\"', quoting=csv.QUOTE_ALL, lineterminator=\"\\n\"\n", + " )\n", + "\n", + " for root, _, files in os.walk(train_data_dir):\n", + " for file in files:\n", + " if file.endswith(\".txt\"):\n", + " class_name = root.split(\"/\")[-1]\n", + " file_fn = os.path.join(root, file)\n", + " with open(file_fn, \"r\") as f:\n", + " content = f.readlines()\n", + " lines = [x.strip().strip('\"') for x in content]\n", + " writer.writerow((lines[0], class_name))\n", + "\n", + " fp.close()\n", + "\n", + " train_gcs_url = upload_blob(\n", + " bucket_name, train_data_fn, os.path.join(destination_blob_prefix, \"train.csv\")\n", + " )\n", + "\n", + " return train_gcs_url" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "q3UDGt8MrKoY" + }, + "source": [ + "### Download Data" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Ks2ahWfUrKoa" + }, + "outputs": [], + "source": [ + "data_dir = download_data(DATA_DIR)\n", + "print(f\"Data is downloaded to: {data_dir}\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "l7AHtjj3rKoa" + }, + "outputs": [], + "source": [ + "!ls $data_dir" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "puroD_Wr-W4Y" + }, + "outputs": [], + "source": [ + "!ls $data_dir/train" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bQNL5AjtrKoa" + }, + "source": [ + "### Prepare CSV Files for Creating Managed Dataset" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "9ewrG6VArKob" + }, + "source": [ + "#### Create CSV Files using Data Content" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "GvbpGr0ArKob" + }, + "outputs": [], + "source": [ + "gcs_source_train_url = upload_train_data_to_gcs(\n", + " train_data_dir=os.path.join(data_dir, \"train\"),\n", + " bucket_name=BUCKET_NAME,\n", + " destination_blob_prefix=f\"{GCS_PREFIX}/data\",\n", + ")\n", + "\n", + "print(f\"Train data content is loaded to {gcs_source_train_url}\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Gg_OiKJUrKoc" + }, + "outputs": [], + "source": [ + "!gsutil ls gs://$BUCKET_NAME/$GCS_PREFIX/data" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "qzKx7Z2WrKoc" + }, + "source": [ + "# Create Custom Training Python Package\n", + "\n", + "Before you can perform custom training with a pre-built container, you must create a [Python Source Distribution](https://docs.python.org/3/distutils/sourcedist.html) that contains your training application and upload it to a Cloud Storage bucket that your Google Cloud project can access.\n", + "\n", + "We will create a directory and write all of our package build artifacts into that folder." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "eBDjNpkLrKoc" + }, + "outputs": [], + "source": [ + "PYTHON_PACKAGE_APPLICATION_DIR = f\"{TASK_NAME}/trainer\"\n", + "\n", + "!mkdir -p $PYTHON_PACKAGE_APPLICATION_DIR\n", + "!touch $PYTHON_PACKAGE_APPLICATION_DIR/__init__.py" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "sPSRvSvMrKod" + }, + "source": [ + "### Write the Training Script" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Zupq7mwzrKod" + }, + "outputs": [], + "source": [ + "%%writefile {PYTHON_PACKAGE_APPLICATION_DIR}/task.py\n", + "\n", + "\n", + "import os\n", + "import argparse\n", + "\n", + "import tensorflow as tf\n", + "from tensorflow.keras import layers\n", + "from tensorflow.keras import losses\n", + "from tensorflow.keras.layers.experimental.preprocessing import TextVectorization\n", + "\n", + "import json\n", + "import tqdm\n", + "\n", + "VOCAB_SIZE = 10000\n", + "MAX_SEQUENCE_LENGTH = 250\n", + "\n", + "def str2bool(v):\n", + " if isinstance(v, bool):\n", + " return v\n", + " if v.lower() in ('yes', 'true', 't', 'y', '1'):\n", + " return True\n", + " elif v.lower() in ('no', 'false', 'f', 'n', '0'):\n", + " return False\n", + " else:\n", + " raise argparse.ArgumentTypeError('Boolean value expected.')\n", + "\n", + "def build_model(num_classes, loss, optimizer, metrics, vectorize_layer):\n", + " # vocab_size is VOCAB_SIZE + 1 since 0 is used additionally for padding.\n", + " model = tf.keras.Sequential([\n", + " vectorize_layer,\n", + " layers.Embedding(VOCAB_SIZE + 1, 64, mask_zero=True),\n", + " layers.Conv1D(64, 5, padding=\"valid\", activation=\"relu\", strides=2),\n", + " layers.GlobalMaxPooling1D(),\n", + " layers.Dense(num_classes),\n", + " layers.Activation('softmax')\n", + " ])\n", + " model.compile(\n", + " loss=loss,\n", + " optimizer=optimizer,\n", + " metrics=metrics)\n", + "\n", + " return model\n", + "\n", + "def get_string_labels(predicted_scores_batch, class_names):\n", + " predicted_labels = tf.argmax(predicted_scores_batch, axis=1)\n", + " predicted_labels = tf.gather(class_names, predicted_labels)\n", + " return predicted_labels\n", + "\n", + "def predict(export_model, class_names, inputs):\n", + " predicted_scores = export_model.predict(inputs)\n", + " predicted_labels = get_string_labels(predicted_scores, class_names)\n", + " return predicted_labels\n", + "\n", + "def parse_args():\n", + " parser = argparse.ArgumentParser(\n", + " description='Keras Text Classification on Stack Overflow Questions')\n", + " parser.add_argument(\n", + " '--epochs', default=25, type=int, help='number of training epochs')\n", + " parser.add_argument(\n", + " '--batch-size', default=16, type=int, help='mini-batch size')\n", + " parser.add_argument(\n", + " '--model-dir', default=os.getenv('AIP_MODEL_DIR'), type=str, help='model directory')\n", + " parser.add_argument(\n", + " '--data-dir', default='./data', type=str, help='data directory')\n", + " parser.add_argument(\n", + " '--test-run', default=False, type=str2bool, help='test run the training application, i.e. 1 epoch for training using sample dataset')\n", + " parser.add_argument(\n", + " '--model-version', default=1, type=int, help='model version')\n", + " args = parser.parse_args()\n", + " return args\n", + "\n", + "def load_aip_dataset(aip_data_uri_pattern, batch_size, class_names, test_run, shuffle=True, seed=42):\n", + "\n", + " data_file_urls = list()\n", + " labels = list()\n", + "\n", + " class_indices = dict(zip(class_names, range(len(class_names))))\n", + " num_classes = len(class_names)\n", + "\n", + " for aip_data_uri in tqdm.tqdm(tf.io.gfile.glob(pattern=aip_data_uri_pattern)):\n", + " with tf.io.gfile.GFile(name=aip_data_uri, mode='r') as gfile:\n", + " for line in gfile.readlines():\n", + " line = json.loads(line)\n", + " data_file_urls.append(line['textContent'])\n", + " classification_annotation = line['classificationAnnotations'][0]\n", + " label = classification_annotation['displayName']\n", + " labels.append(class_indices[label])\n", + " if test_run:\n", + " break\n", + "\n", + " data = list()\n", + " for data_file_url in tqdm.tqdm(data_file_urls):\n", + " with tf.io.gfile.GFile(name=data_file_url, mode='r') as gf:\n", + " txt = gf.read()\n", + " data.append(txt)\n", + "\n", + " print(f' data files count: {len(data_file_urls)}')\n", + " print(f' data count: {len(data)}')\n", + " print(f' labels count: {len(labels)}')\n", + "\n", + " dataset = tf.data.Dataset.from_tensor_slices(data)\n", + " label_ds = tf.data.Dataset.from_tensor_slices(labels)\n", + " label_ds = label_ds.map(lambda x: tf.one_hot(x, num_classes))\n", + "\n", + " dataset = tf.data.Dataset.zip((dataset, label_ds))\n", + "\n", + " if shuffle:\n", + " # Shuffle locally at each iteration\n", + " dataset = dataset.shuffle(buffer_size=batch_size * 8, seed=seed)\n", + " dataset = dataset.batch(batch_size)\n", + " # Users may need to reference `class_names`.\n", + " dataset.class_names = class_names\n", + "\n", + " return dataset\n", + "\n", + "def main():\n", + "\n", + " args = parse_args()\n", + "\n", + " class_names = ['csharp', 'java', 'javascript', 'python']\n", + " class_indices = dict(zip(class_names, range(len(class_names))))\n", + " num_classes = len(class_names)\n", + " print(f' class names: {class_names}')\n", + " print(f' class indices: {class_indices}')\n", + " print(f' num classes: {num_classes}')\n", + "\n", + " epochs = 1 if args.test_run else args.epochs\n", + "\n", + " aip_model_dir = os.environ.get('AIP_MODEL_DIR')\n", + " aip_data_format = os.environ.get('AIP_DATA_FORMAT')\n", + " aip_training_data_uri = os.environ.get('AIP_TRAINING_DATA_URI')\n", + " aip_validation_data_uri = os.environ.get('AIP_VALIDATION_DATA_URI')\n", + " aip_test_data_uri = os.environ.get('AIP_TEST_DATA_URI')\n", + "\n", + " print(f\"aip_model_dir: {aip_model_dir}\")\n", + " print(f\"aip_data_format: {aip_data_format}\")\n", + " print(f\"aip_training_data_uri: {aip_training_data_uri}\")\n", + " print(f\"aip_validation_data_uri: {aip_validation_data_uri}\")\n", + " print(f\"aip_test_data_uri: {aip_test_data_uri}\")\n", + "\n", + " print('Loading AIP dataset')\n", + " train_ds = load_aip_dataset(\n", + " aip_training_data_uri, args.batch_size, class_names, args.test_run)\n", + " print('AIP training dataset is loaded')\n", + " val_ds = load_aip_dataset(\n", + " aip_validation_data_uri, 1, class_names, args.test_run)\n", + " print('AIP validation dataset is loaded')\n", + " test_ds = load_aip_dataset(\n", + " aip_test_data_uri, 1, class_names, args.test_run)\n", + " print('AIP test dataset is loaded')\n", + "\n", + " vectorize_layer = TextVectorization(\n", + " max_tokens=VOCAB_SIZE,\n", + " output_mode='int',\n", + " output_sequence_length=MAX_SEQUENCE_LENGTH)\n", + "\n", + " train_text = train_ds.map(lambda text, labels: text)\n", + " vectorize_layer.adapt(train_text)\n", + " print('The vectorize_layer is adapted')\n", + "\n", + "\n", + " print('Build model')\n", + " optimizer = 'adam'\n", + " metrics = ['accuracy']\n", + "\n", + " model = build_model(\n", + " num_classes, losses.CategoricalCrossentropy(from_logits=True), optimizer, metrics, vectorize_layer)\n", + "\n", + " history = model.fit(train_ds, validation_data=val_ds, epochs=epochs)\n", + " history = history.history\n", + "\n", + " print('Training accuracy: {acc}, loss: {loss}'.format(\n", + " acc=history['accuracy'][-1], loss=history['loss'][-1]))\n", + " print('Validation accuracy: {acc}, loss: {loss}'.format(\n", + " acc=history['val_accuracy'][-1], loss=history['val_loss'][-1]))\n", + "\n", + " loss, accuracy = model.evaluate(test_ds)\n", + " print('Test accuracy: {acc}, loss: {loss}'.format(\n", + " acc=accuracy, loss=loss))\n", + "\n", + " inputs = [\n", + " \"how do I extract keys from a dict into a list?\", # python\n", + " \"debug public static void main(string[] args) {...}\", # java\n", + " ]\n", + " predicted_labels = predict(model, class_names, inputs)\n", + " for input, label in zip(inputs, predicted_labels):\n", + " print(f'Question: {input}')\n", + " print(f'Predicted label: {label.numpy()}')\n", + "\n", + " model_export_path = os.path.join(args.model_dir, str(args.model_version))\n", + " model.save(model_export_path)\n", + " print(f'Model version {args.model_version} is exported to {args.model_dir}')\n", + "\n", + " loaded = tf.saved_model.load(model_export_path)\n", + " input_name = list(loaded.signatures['serving_default'].structured_input_signature[1].keys())[0]\n", + " print(f'Serving function input: {input_name}')\n", + "\n", + " return\n", + "\n", + "if __name__ == '__main__':\n", + " main()\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "JAhynoNBrKoh" + }, + "source": [ + "### Build Package" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "kXSBQcqdrKof" + }, + "outputs": [], + "source": [ + "%%writefile {TASK_DIR}/setup.py\n", + "\n", + "from setuptools import find_packages\n", + "from setuptools import setup\n", + "\n", + "setup(\n", + " name='trainer',\n", + " version='0.1',\n", + " packages=find_packages(),\n", + " install_requires=(),\n", + " include_package_data=True,\n", + " description='My training application.'\n", + ")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "MclanW0UrKoh" + }, + "outputs": [], + "source": [ + "!ls $TASK_DIR" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "j19Nk2vzrKoh" + }, + "outputs": [], + "source": [ + "!cd $TASK_DIR && python3 setup.py sdist --formats=gztar" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "9c1r_k69rKoi" + }, + "outputs": [], + "source": [ + "!ls -ltr $TASK_DIR/dist/trainer-0.1.tar.gz" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "zyQv9AcNrKoi" + }, + "source": [ + "### Upload the Package to GCS" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "9WjF8SFrrKoi" + }, + "outputs": [], + "source": [ + "destination_blob_name = f\"custom-training-python-package/{APP_NAME}/trainer-0.1.tar.gz\"\n", + "source_file_name = f\"{TASK_DIR}/dist/trainer-0.1.tar.gz\"\n", + "\n", + "python_package_gcs_uri = upload_blob(\n", + " BUCKET_NAME, source_file_name, destination_blob_name\n", + ")\n", + "python_module_name = \"trainer.task\"\n", + "\n", + "print(f\"Custom Training Python Package is uploaded to: {python_package_gcs_uri}\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "hWJpv4ejrWi0" + }, + "source": [ + "# Create TensorFlow Serving Container" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "IMOad4Z4aOgv" + }, + "source": [ + "Download the TensorFlow Serving Docker image." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "KC8e1yrHrVyA" + }, + "outputs": [], + "source": [ + "!docker pull tensorflow/serving:latest" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "rtiNicSyaTk1" + }, + "source": [ + "Create a tag for registering the image and register the image with Cloud Container Registry (gcr.io)." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "sUfgSddcsRvw" + }, + "outputs": [], + "source": [ + "TF_SERVING_CONTAINER_IMAGE_URI = f\"gcr.io/{MY_PROJECT}/tf-serving\"" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Bw_A7ynTsjTF" + }, + "outputs": [], + "source": [ + "!docker tag tensorflow/serving $TF_SERVING_CONTAINER_IMAGE_URI\n", + "!docker push $TF_SERVING_CONTAINER_IMAGE_URI" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Rc_MIJwlrKoj" + }, + "source": [ + "# Run Custom Python Package Training with Managed Text Dataset" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "eiNL0HSVrKoj" + }, + "source": [ + "## Initialize Vertex SDK for Python\n", + "\n", + "Initialize the *client* for Vertex AI." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "J5MqvGtMrKoj" + }, + "outputs": [], + "source": [ + "from google.cloud import aiplatform\n", + "\n", + "aiplatform.init(project=MY_PROJECT, staging_bucket=MY_STAGING_BUCKET)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "KGT1uT-HrKoj" + }, + "source": [ + "## Create a Dataset on Vertex AI\n", + "We will now create a Vertex AI text dataset using the previously prepared csv files. Choose one of the options below. " + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "lY566K2crKok" + }, + "outputs": [], + "source": [ + "dataset_display_name = f\"temp-{APP_NAME}-content\"\n", + "gcs_source = gcs_source_train_url" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "kyLoUsx9rKok" + }, + "source": [ + "#### Option 1: Create a Dataset with CSV File" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "h7QwC1ThrKok" + }, + "outputs": [], + "source": [ + "dataset = aiplatform.TextDataset.create(\n", + " display_name=dataset_display_name,\n", + " gcs_source=gcs_source,\n", + " import_schema_uri=aiplatform.schema.dataset.ioformat.text.single_label_classification,\n", + " sync=False,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "xksBj3M0rKok" + }, + "source": [ + "#### Option 2: Create a Dataset, then Import CSV File" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "93w8LGlx7uWy" + }, + "source": [ + "```\n", + "dataset = aiplatform.TextDataset.create(\n", + " display_name=dataset_display_name,\n", + ")\n", + "dataset.import_data(\n", + " gcs_source=gcs_source, \n", + " import_schema_uri=aiplatform.schema.dataset.ioformat.text.single_label_classification,\n", + " sync=False\n", + ")\n", + "```" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ncP9QwvkrKok" + }, + "source": [ + "#### Option 3: Retrieve a Dataset on Vertex AI\n", + "If you have previously created a Dataset on Vertex AI, you can retrieve the dataset using the `dataset_name`." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "OmiQDoS_7ncz" + }, + "source": [ + "```\n", + "dataset_name = 'YOUR DATASET NAME'\n", + "\n", + "dataset = aiplatform.TextDataset(dataset_name)\n", + "dataset.resource_name\n", + "```" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "NzLvubIrrKon" + }, + "source": [ + "## Launch a Training Job and Create a Model on Vertex AI\n", + "\n", + "We will now train a model with the python package we just built." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Cyz86ixRYNtT" + }, + "source": [ + "### Config a Training Job" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "9zk8SRuK1Xvi" + }, + "outputs": [], + "source": [ + "MODEL_NAME = APP_NAME\n", + "PRE_BUILT_TRAINING_CONTAINER_IMAGE_URI = (\n", + " \"gcr.io/cloud-aiplatform/training/tf-cpu.2-3:latest\"\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "rxc_yVxkbbbd" + }, + "source": [ + "You will need to specify the python package that was built and uploaded to GCS, the module name of the python package, the pre-built training container image uri for training, and in this example, we are using TensorFlow serving container for prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "uRGrFdxOftD1" + }, + "outputs": [], + "source": [ + "job = aiplatform.CustomPythonPackageTrainingJob(\n", + " display_name=f\"temp_{TASK_NAME}_tf-serving\",\n", + " python_package_gcs_uri=python_package_gcs_uri,\n", + " python_module_name=python_module_name,\n", + " container_uri=PRE_BUILT_TRAINING_CONTAINER_IMAGE_URI,\n", + " model_serving_container_image_uri=TF_SERVING_CONTAINER_IMAGE_URI,\n", + " model_serving_container_command=[\"/usr/bin/tensorflow_model_server\"],\n", + " model_serving_container_args=[\n", + " f\"--model_name={MODEL_NAME}\",\n", + " \"--model_base_path=$(AIP_STORAGE_URI)\",\n", + " \"--rest_api_port=8080\",\n", + " \"--port=8500\",\n", + " \"--file_system_poll_wait_seconds=31540000\",\n", + " ],\n", + " model_serving_container_predict_route=f\"/v1/models/{MODEL_NAME}:predict\",\n", + " model_serving_container_health_route=f\"/v1/models/{MODEL_NAME}\",\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "JPfZGlZeTN72" + }, + "source": [ + "### Run the Training Job" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "V6iY5wcerKon" + }, + "outputs": [], + "source": [ + "model = job.run(\n", + " dataset=dataset,\n", + " annotation_schema_uri=aiplatform.schema.dataset.annotation.text.classification,\n", + " args=[\"--epochs\", \"50\"],\n", + " replica_count=1,\n", + " model_display_name=f\"temp_{TASK_NAME}_tf-serving\",\n", + " sync=False,\n", + ")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "zV1PjANLrKoo" + }, + "outputs": [], + "source": [ + "model.wait()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "XO0-zdXxrKoo" + }, + "source": [ + "# Deploy a Model and Create an Endpoint on Vertex AI\n", + "\n", + "Deploy your model, then wait until the model FINISHES deployment before proceeding to prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "tEg2IDwPftD2" + }, + "outputs": [], + "source": [ + "endpoint = model.deploy(machine_type=\"n1-standard-4\", sync=False)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "9AoIJdCHrKop" + }, + "outputs": [], + "source": [ + "endpoint.wait()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Z87MXJEDrKoq" + }, + "source": [ + "## Predict on the Endpoint" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "vQXGCg8IrKoq" + }, + "outputs": [], + "source": [ + "class_names = [\"csharp\", \"java\", \"javascript\", \"python\"]\n", + "\n", + "class_ids = range(len(class_names))\n", + "\n", + "class_indices = dict(zip(class_names, class_ids))\n", + "class_maps = dict(zip(class_ids, class_names))\n", + "print(f\"Class Indices: {class_indices}\")\n", + "print(f\"Class Maps: {class_maps}\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Di8RtRxipm63" + }, + "outputs": [], + "source": [ + "text_inputs = [\n", + " \"how do I extract keys from a dict into a list?\", # python\n", + " \"debug public static void main(string[] args) {...}\", # java\n", + "]" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "dPHYGAlvlTAj" + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "\n", + "predictions = endpoint.predict(instances=[[text] for text in text_inputs])\n", + "for text, predicted_scores in zip(text_inputs, predictions.predictions):\n", + " class_id = np.argmax(predicted_scores)\n", + " class_name = class_maps[class_id]\n", + " print(f\"Question: {text}\")\n", + " print(f\"Predicted Tag: {class_name}\\n\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gRkvzr2-LgA9" + }, + "source": [ + "# Batch Prediction Job on the Model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "nlzRioF7UGFd" + }, + "outputs": [], + "source": [ + "import json\n", + "\n", + "import tensorflow as tf\n", + "\n", + "\n", + "def upload_test_data_to_gcs(test_data_dir, test_gcs_url):\n", + " \"\"\"Create JSON file using test data content.\"\"\"\n", + "\n", + " input_name = \"text_vectorization_input\"\n", + "\n", + " with tf.io.gfile.GFile(test_gcs_url, \"w\") as gf:\n", + "\n", + " for root, _, files in os.walk(test_data_dir):\n", + " for file in files:\n", + " if file.endswith(\".txt\"):\n", + " file_fn = os.path.join(root, file)\n", + " with open(file_fn, \"r\") as f:\n", + " content = f.readlines()\n", + " lines = [x.strip().strip('\"') for x in content]\n", + "\n", + " data = {input_name: [lines[0]]}\n", + " gf.write(json.dumps(data))\n", + " gf.write(\"\\n\")\n", + " return" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "C_sbOm-5ud5C" + }, + "outputs": [], + "source": [ + "gcs_source_test_url = f\"gs://{BUCKET_NAME}/{GCS_PREFIX}/data/test.json\"\n", + "upload_test_data_to_gcs(\n", + " test_data_dir=os.path.join(data_dir, \"test\"), test_gcs_url=gcs_source_test_url\n", + ")\n", + "\n", + "print(f\"Test data content is loaded to {gcs_source_test_url}\")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "QBMBk2WxLqBP" + }, + "outputs": [], + "source": [ + "!gsutil ls $gcs_source_test_url" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "2JhoTiD5LuDW" + }, + "outputs": [], + "source": [ + "batch_predict_job = model.batch_predict(\n", + " job_display_name=f\"temp_{TASK_NAME}_tf-serving\",\n", + " gcs_source=gcs_source_test_url,\n", + " gcs_destination_prefix=f\"gs://{BUCKET_NAME}/{GCS_PREFIX}/batch_prediction\",\n", + " machine_type=\"n1-standard-4\",\n", + " sync=False,\n", + ")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "retsI3LLls_W" + }, + "outputs": [], + "source": [ + "batch_predict_job.wait()\n", + "bp_iter_outputs = batch_predict_job.iter_outputs()\n", + "\n", + "prediction_errors_stats = list()\n", + "prediction_results = list()\n", + "for blob in bp_iter_outputs:\n", + " if blob.name.split(\"/\")[-1].startswith(\"prediction.errors_stats\"):\n", + " prediction_errors_stats.append(blob.name)\n", + " if blob.name.split(\"/\")[-1].startswith(\"prediction.results\"):\n", + " prediction_results.append(blob.name)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "KmfK3Tzzlv9C" + }, + "outputs": [], + "source": [ + "tags = list()\n", + "for prediction_result in prediction_results:\n", + " gfile_name = f\"gs://{bp_iter_outputs.bucket.name}/{prediction_result}\"\n", + " with tf.io.gfile.GFile(name=gfile_name, mode=\"r\") as gfile:\n", + " for line in gfile.readlines():\n", + " line = json.loads(line)\n", + " text = line[\"instance\"][\"text_vectorization_input\"][0]\n", + " prediction = line[\"prediction\"]\n", + " class_id = np.argmax(prediction)\n", + " class_name = class_maps[class_id]\n", + " tags.append([text, class_name])" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "UcMQ-daLn_oh" + }, + "outputs": [], + "source": [ + "import pandas as pd\n", + "\n", + "tags_df = pd.DataFrame(tags, columns=[\"question\", \"tag\"])\n", + "tags_df.head()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "QUV_PHjtoCpQ" + }, + "outputs": [], + "source": [ + "tags_df[\"tag\"].value_counts()" + ] + } + ], + "metadata": { + "colab": { + "collapsed_sections": [ + "bcKMGrqDrKoS", + "q3UDGt8MrKoY", + "bQNL5AjtrKoa", + "9ewrG6VArKob", + "eiNL0HSVrKoj", + "KGT1uT-HrKoj", + "kyLoUsx9rKok", + "xksBj3M0rKok", + "ncP9QwvkrKok", + "Cyz86ixRYNtT" + ], + "name": "AI_Platform_(Unified)_SDK_Custom_Training_Python_Package_Managed_Text_Dataset_Tensorflow_Serving_Container.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/sdk/SDK_Custom_Training_with_Unmanaged_Image_Dataset.ipynb b/notebooks/community/sdk/SDK_Custom_Training_with_Unmanaged_Image_Dataset.ipynb new file mode 100644 index 000000000..7660cd7ce --- /dev/null +++ b/notebooks/community/sdk/SDK_Custom_Training_with_Unmanaged_Image_Dataset.ipynb @@ -0,0 +1,396 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ur8xi4C7S06n" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "1c955a52adac" + }, + "source": [ + "# Feedback or issues?\n", + "For any feedback or questions, please open an [issue](https://github.com/googleapis/python-aiplatform/issues)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "eHLV0D7Y5jtU" + }, + "source": [ + "# Vertex SDK for Python: Custom Training Example with Unmanaged Image Dataset\n", + "\n", + "\n", + "To use this Colaboratory notebook, you copy the notebook to your own Google Drive and open it with Colaboratory (or Colab). You can run each step, or cell, and see its results. To run a cell, use Shift+Enter. Colab automatically displays the return value of the last line in each cell. For more information about running notebooks in Colab, see the [Colab welcome page](https://colab.research.google.com/notebooks/welcome.ipynb).\n", + "\n", + "This notebook demonstrate how to create a custom model based on an image dataset. It will require you provide a bucket where the dataset will be stored.\n", + "\n", + "Note: you may incur charges for training, prediction, storage or usage of other GCP products in connection with testing this SDK." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "lld3eeJUs5yM" + }, + "source": [ + "# Install Vertex SDK for Python, Authenticate, and upload of a Dataset to your GCS bucket\n", + "\n", + "\n", + "After the SDK installation the kernel will be automatically restarted. You may see this error message `Your session crashed for an unknown reason` which is normal." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "sBfZtR4X1Dr_" + }, + "outputs": [], + "source": [ + "!pip3 uninstall -y google-cloud-aiplatform\n", + "!pip3 install google-cloud-aiplatform\n", + "import IPython\n", + "\n", + "app = IPython.Application.instance()\n", + "app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "c0SNmTBeD2nV" + }, + "source": [ + "### Enter your project and GCS bucket\n", + "\n", + "Enter your Project Id in the cell below. Then run the cell to make sure the Cloud SDK uses the right project for all the commands in this notebook." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "f5be5dce7259" + }, + "outputs": [], + "source": [ + "import sys\n", + "\n", + "if \"google.colab\" in sys.modules:\n", + " from google.colab import auth\n", + "\n", + " auth.authenticate_user()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "kng9iKBwqcS5" + }, + "outputs": [], + "source": [ + "MY_PROJECT = \"YOUR PROJECT\"\n", + "MY_STAGING_BUCKET = \"gs://YOUR BUCKET\" # bucket should be in same region as ucaip" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "rk43VP_IqcTE" + }, + "source": [ + "# Initialize Vertex SDK for Python\n", + "\n", + "Initialize the *client* for Vertex AI" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "VCiC9gBWqcTF" + }, + "outputs": [], + "source": [ + "from google.cloud import aiplatform\n", + "\n", + "aiplatform.init(project=MY_PROJECT, staging_bucket=MY_STAGING_BUCKET)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "VcEOYYolqcTN" + }, + "source": [ + "# Write your Training Script\n", + "- Write this cell as a file which will be used for custom training.\n", + "- Instead of using a managed dataset, a Tensorflow Dataset URI is passed in through the 'args' parameter of the 'run' function. The script will download the data from the URI at training time." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "78SHpd0tt8UQ" + }, + "outputs": [], + "source": [ + "%%writefile training_script.py\n", + "\n", + "# Source: https://cloud.google.com/vertex-ai/docs/tutorials/image-recognition-custom\n", + "\n", + "import argparse\n", + "import logging\n", + "import os\n", + "\n", + "import tensorflow as tf\n", + "import tensorflow_datasets as tfds\n", + "\n", + "IMG_WIDTH = 128\n", + "\n", + "def normalize_img(image):\n", + " \"\"\"Normalizes image.\n", + "\n", + " * Resizes image to IMG_WIDTH x IMG_WIDTH pixels\n", + " * Casts values from `uint8` to `float32`\n", + " * Scales values from [0, 255] to [0, 1]\n", + "\n", + " Returns:\n", + " A tensor with shape (IMG_WIDTH, IMG_WIDTH, 3). (3 color channels)\n", + " \"\"\"\n", + " image = tf.image.resize_with_pad(image, IMG_WIDTH, IMG_WIDTH)\n", + " return image / 255.\n", + "\n", + "\n", + "def normalize_img_and_label(image, label):\n", + " \"\"\"Normalizes image and label.\n", + "\n", + " * Performs normalize_img on image\n", + " * Passes through label unchanged\n", + "\n", + " Returns:\n", + " Tuple (image, label) where\n", + " * image is a tensor with shape (IMG_WIDTH, IMG_WIDTH, 3). (3 color\n", + " channels)\n", + " * label is an unchanged integer [0, 4] representing flower type\n", + " \"\"\"\n", + " return normalize_img(image), label\n", + "\n", + "def get_args():\n", + " \"\"\"Argument parser.\n", + " Returns:\n", + " Dictionary of arguments.\n", + " \"\"\"\n", + " parser = argparse.ArgumentParser(description='Flower classification sample')\n", + " parser.add_argument(\n", + " '--tfds',\n", + " default=None,\n", + " help='The tfds URI from https://www.tensorflow.org/datasets/ to load the data from')\n", + "\n", + " args = parser.parse_args()\n", + " return args\n", + "\n", + "# Training settings\n", + "args = get_args()\n", + "\n", + "if 'AIP_MODEL_DIR' not in os.environ:\n", + " raise KeyError(\n", + " 'The `AIP_MODEL_DIR` environment variable has not been' +\n", + " 'set. See https://cloud.google.com/vertex-ai/docs/tutorials/image-recognition-custom/training'\n", + " )\n", + "output_directory = os.environ['AIP_MODEL_DIR']\n", + "\n", + "logging.info('Loading and preprocessing data ...')\n", + "dataset = tfds.load(args.tfds,\n", + " split='train',\n", + " try_gcs=True,\n", + " shuffle_files=True,\n", + " as_supervised=True)\n", + "dataset = dataset.map(normalize_img_and_label,\n", + " num_parallel_calls=tf.data.experimental.AUTOTUNE)\n", + "dataset = dataset.cache()\n", + "dataset = dataset.shuffle(1000)\n", + "dataset = dataset.batch(128)\n", + "dataset = dataset.prefetch(tf.data.experimental.AUTOTUNE)\n", + "\n", + "logging.info('Creating and training model ...')\n", + "model = tf.keras.Sequential([\n", + " tf.keras.layers.Conv2D(16,\n", + " 3,\n", + " padding='same',\n", + " activation='relu',\n", + " input_shape=(IMG_WIDTH, IMG_WIDTH, 3)),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Conv2D(32, 3, padding='same', activation='relu'),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Conv2D(64, 3, padding='same', activation='relu'),\n", + " tf.keras.layers.MaxPooling2D(),\n", + " tf.keras.layers.Flatten(),\n", + " tf.keras.layers.Dense(512, activation=\"relu\"),\n", + " tf.keras.layers.Dense(5) # 5 classes\n", + "])\n", + "model.compile(\n", + " optimizer='adam',\n", + " loss=tf.keras.losses.SparseCategoricalCrossentropy(from_logits=True),\n", + " metrics=['accuracy'])\n", + "model.fit(dataset, epochs=10)\n", + "\n", + "logging.info(f'Exporting SavedModel to: {output_directory}')\n", + "# Add softmax layer for intepretability\n", + "probability_model = tf.keras.Sequential([model, tf.keras.layers.Softmax()])\n", + "probability_model.save(output_directory)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "6-bBqipfqcTS" + }, + "source": [ + "# Launch a Training Job to Create a Model\n", + "\n", + "Once we have defined your training script, we will create a model." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "btb6d48lqcTT" + }, + "outputs": [], + "source": [ + "job = aiplatform.CustomTrainingJob(\n", + " display_name=\"train-flowers-dist-1-replica\",\n", + " script_path=\"training_script.py\",\n", + " container_uri=\"gcr.io/cloud-aiplatform/training/tf-cpu.2-2:latest\",\n", + " requirements=[\"gcsfs==0.7.1\"],\n", + " model_serving_container_image_uri=\"gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest\",\n", + ")\n", + "model = job.run(\n", + " args=[\"--tfds\", \"tf_flowers:3.*.*\"],\n", + " replica_count=1,\n", + " model_display_name=\"flowers-model\",\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "5vhDsMJNqcTW" + }, + "source": [ + "# Deploy Your Model\n", + "\n", + "Deploy your model, then wait until the model FINISHES deployment before proceeding to prediction.\n" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Y9GH72wWqcTX" + }, + "outputs": [], + "source": [ + "endpoint = model.deploy(machine_type=\"n1-standard-4\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "nIw1ifPuqcTb" + }, + "source": [ + "# Predict on the Endpoint\n", + "To do a prediction you will need some flowers images. You can download some photos of flowers or use the ones provided below." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wn8fUtq389r4" + }, + "outputs": [], + "source": [ + "!gsutil -m cp -R gs://cloud-ml-data/img/flower_photos/daisy/14221848160_7f0a37c395.jpg .\n", + "!gsutil -m cp -R gs://cloud-ml-data/img/flower_photos/tulips/13289268363_b9337d751e.jpg .\n", + "!gsutil -m cp -R gs://cloud-ml-data/img/flower_photos/sunflowers/14623719696_1bb7970208_n.jpg ." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "_ktakP9r7mCt" + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "from PIL import Image\n", + "\n", + "daisy_floats = np.array(Image.open(\"14221848160_7f0a37c395.jpg\"))" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "xzGidweY8Dgy" + }, + "outputs": [], + "source": [ + "small_image = np.array(Image.fromarray(np.uint8(daisy_floats)).resize((128, 128)))" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cL6VdVufgH8a" + }, + "outputs": [], + "source": [ + "endpoint.predict(instances=[small_image.tolist()])" + ] + } + ], + "metadata": { + "colab": { + "collapsed_sections": [], + "name": "AI_Platform_(Unified)_SDK_Custom_Training_with_Unmanaged_Image_Dataset.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/sdk/SDK_End_to_End_Tabular_Custom_Training.ipynb b/notebooks/community/sdk/SDK_End_to_End_Tabular_Custom_Training.ipynb new file mode 100644 index 000000000..a9456d10d --- /dev/null +++ b/notebooks/community/sdk/SDK_End_to_End_Tabular_Custom_Training.ipynb @@ -0,0 +1,335 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ur8xi4C7S06n" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "1c955a52adac" + }, + "source": [ + "# Feedback or issues?\n", + "For any feedback or questions, please open an [issue](https://github.com/googleapis/python-aiplatform/issues)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "eHLV0D7Y5jtU" + }, + "source": [ + "# Vertex SDK for Python: Custom Tabular Model Training Example\n", + "\n", + "To use this Colaboratory notebook, you copy the notebook to your own Google Drive and open it with Colaboratory (or Colab). You can run each step, or cell, and see its results. To run a cell, use Shift+Enter. Colab automatically displays the return value of the last line in each cell. For more information about running notebooks in Colab, see the [Colab welcome page](https://colab.research.google.com/notebooks/welcome.ipynb).\n", + "\n", + "This notebook demonstrate how to create a custom model based on a tabular dataset. It will require you provide a bucket where the dataset CSV will be stored.\n", + "\n", + "Note: you may incur charges for training, prediction, storage or usage of other GCP products in connection with testing this SDK." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "lld3eeJUs5yM" + }, + "source": [ + "# Install Vertex SDK for Python, Authenticate, and upload of a Dataset to your GCS bucket\n", + "\n", + "\n", + "After the SDK installation the kernel will be automatically restarted. You may see this error message `Your session crashed for an unknown reason` which is normal." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "sBfZtR4X1Dr_" + }, + "outputs": [], + "source": [ + "!pip3 uninstall -y google-cloud-aiplatform\n", + "!pip3 install google-cloud-aiplatform\n", + "import IPython\n", + "\n", + "app = IPython.Application.instance()\n", + "app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "f45840a3cff3" + }, + "outputs": [], + "source": [ + "import sys\n", + "\n", + "if \"google.colab\" in sys.modules:\n", + " from google.colab import auth\n", + "\n", + " auth.authenticate_user()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "c0SNmTBeD2nV" + }, + "source": [ + "### Enter your project and GCS bucket\n", + "\n", + "Enter your Project Id in the cell below. Then run the cell to make sure the Cloud SDK uses the right project for all the commands in this notebook." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "kng9iKBwqcS5" + }, + "outputs": [], + "source": [ + "MY_PROJECT = \"YOUR PROJECT\"\n", + "MY_STAGING_BUCKET = \"gs://YOUR BUCKET\" # bucket should be in same region as Vertex AI" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "9nokDKBAxwV8" + }, + "source": [ + "The dataset we are using is the Abalone Dataset. For more information about this dataset please visit: https://archive.ics.uci.edu/ml/datasets/abalone" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "V_T10yTTqcS_" + }, + "outputs": [], + "source": [ + "!wget https://storage.googleapis.com/download.tensorflow.org/data/abalone_train.csv\n", + "!gsutil cp abalone_train.csv {MY_STAGING_BUCKET}/data/\n", + "\n", + "gcs_csv_path = f\"{MY_STAGING_BUCKET}/data/abalone_train.csv\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "rk43VP_IqcTE" + }, + "source": [ + "# Initialize Vertex SDK for Python\n", + "\n", + "Initialize the *client* for Vertex AI" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "VCiC9gBWqcTF" + }, + "outputs": [], + "source": [ + "from google.cloud import aiplatform\n", + "\n", + "aiplatform.init(project=MY_PROJECT, staging_bucket=MY_STAGING_BUCKET)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "35QVNhACqcTJ" + }, + "source": [ + "# Create a Managed Tabular Dataset from CSV\n", + "\n", + "A Managed dataset can be used to create an AutoML model or a custom model. " + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "4OfCqaYRqcTJ" + }, + "outputs": [], + "source": [ + "ds = aiplatform.TabularDataset.create(display_name=\"abalone\", gcs_source=[gcs_csv_path])\n", + "\n", + "ds.resource_name" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "VcEOYYolqcTN" + }, + "source": [ + "# Write Training Script\n", + "- Write this cell as a file which will be used for custom training." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "OauJqJmJqcTO" + }, + "outputs": [], + "source": [ + "%%writefile training_script.py\n", + "\n", + "import pandas as pd\n", + "import os\n", + "import tensorflow as tf\n", + "from tensorflow import keras\n", + "from tensorflow.keras import layers\n", + "\n", + "\n", + "# uncomment and bump up replica_count for distributed training\n", + "# strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "# tf.distribute.experimental_set_strategy(strategy)\n", + "\n", + "col_names = [\"Length\", \"Diameter\", \"Height\", \"Whole weight\", \"Shucked weight\", \"Viscera weight\", \"Shell weight\", \"Age\"]\n", + "target = \"Age\"\n", + "\n", + "def aip_data_to_dataframe(wild_card_path):\n", + " return pd.concat([pd.read_csv(fp.numpy().decode(), names=col_names)\n", + " for fp in tf.data.Dataset.list_files([wild_card_path])])\n", + "\n", + "def get_features_and_labels(df):\n", + " return df.drop(target, axis=1).values, df[target].values\n", + "\n", + "def data_prep(wild_card_path):\n", + " return get_features_and_labels(aip_data_to_dataframe(wild_card_path))\n", + "\n", + "\n", + "model = tf.keras.Sequential([layers.Dense(64), layers.Dense(1)])\n", + "model.compile(loss='mse', optimizer='adam')\n", + "\n", + "model.fit(*data_prep(os.environ[\"AIP_TRAINING_DATA_URI\"]),\n", + " epochs=10 ,\n", + " validation_data=data_prep(os.environ[\"AIP_VALIDATION_DATA_URI\"]))\n", + "print(model.evaluate(*data_prep(os.environ[\"AIP_TEST_DATA_URI\"])))\n", + "\n", + "# save as Vertex AI Managed model\n", + "tf.saved_model.save(model, os.environ[\"AIP_MODEL_DIR\"])" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "6-bBqipfqcTS" + }, + "source": [ + "# Launch a Training Job to Create a Model\n", + "\n", + "Once we have defined your training script, we will create a model." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "btb6d48lqcTT" + }, + "outputs": [], + "source": [ + "job = aiplatform.CustomTrainingJob(\n", + " display_name=\"train-abalone-dist-1-replica\",\n", + " script_path=\"training_script.py\",\n", + " container_uri=\"gcr.io/cloud-aiplatform/training/tf-cpu.2-2:latest\",\n", + " requirements=[\"gcsfs==0.7.1\"],\n", + " model_serving_container_image_uri=\"gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest\",\n", + ")\n", + "model = job.run(ds, replica_count=1, model_display_name=\"abalone-model\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "5vhDsMJNqcTW" + }, + "source": [ + "# Deploy Model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Y9GH72wWqcTX" + }, + "outputs": [], + "source": [ + "endpoint = model.deploy(machine_type=\"n1-standard-4\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "nIw1ifPuqcTb" + }, + "source": [ + "# Predict on Endpoint" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "3k6-rSZPqcTc" + }, + "outputs": [], + "source": [ + "prediction = endpoint.predict(\n", + " [\n", + " [0.435, 0.335, 0.11, 0.33399999999999996, 0.1355, 0.0775, 0.0965],\n", + " [0.585, 0.45, 0.125, 0.874, 0.3545, 0.2075, 0.225],\n", + " ]\n", + ")\n", + "prediction" + ] + } + ], + "metadata": { + "colab": { + "collapsed_sections": [], + "name": "AI_Platform_(Unified)_SDK_End_to_End_Tabular_Custom_Training.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/sdk/SDK_Explainable_AI_Custom_Tabular.ipynb b/notebooks/community/sdk/SDK_Explainable_AI_Custom_Tabular.ipynb new file mode 100644 index 000000000..47daff74b --- /dev/null +++ b/notebooks/community/sdk/SDK_Explainable_AI_Custom_Tabular.ipynb @@ -0,0 +1,608 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Snm88Z0sROQB" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "ThTHoy7DRSye" + }, + "source": [ + "# Feedback or issues?\n", + "\n", + "For any feedback or questions, please open an [issue](https://github.com/googleapis/python-aiplatform/issues)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "acg-u47CJ7-T" + }, + "source": [ + "# Explainable AI via MB SDK on Custom Tabular model\n", + "\n", + "To use this Jupyter notebook, copy the notebook to a Google Cloud Notebooks instance with Tensorflow installed and open it. You can run each step, or cell, and see its results. To run a cell, use Shift+Enter. Jupyter automatically displays the return value of the last line in each cell. For more information about running notebooks in Google Cloud Notebook, see the [Google Cloud Notebook guide](https://cloud.google.com/vertex-ai/docs/general/notebooks).\n", + "\n", + "\n", + "This notebook demonstrate how to create an Custom Tabular Model and how to serve the model for online prediction with Explainability.\n", + "\n", + "Note: you may incur charges for training, prediction, storage or usage of other GCP products in connection with testing this SDK." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "-VL5avdUJ7-V" + }, + "source": [ + "### Install Vertex SDK for Python\n", + "\n", + "\n", + "After the SDK installation the kernel will be automatically restarted." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "5-6l0OdRJ7-V" + }, + "outputs": [], + "source": [ + "%%capture\n", + "!pip3 uninstall -y google-cloud-aiplatform\n", + "!pip3 install -y google-cloud-aiplatform tabulate\n", + "import IPython\n", + "\n", + "app = IPython.Application.instance()\n", + "app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "KjLd9x994wm3" + }, + "outputs": [], + "source": [ + "import sys\n", + "\n", + "if \"google.colab\" in sys.modules:\n", + " from google.colab import auth\n", + "\n", + " auth.authenticate_user()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "amZxHeNzRhgN" + }, + "source": [ + "### Enter Your Project and GCS Bucket\n", + "\n", + "Enter your Project Id in the cell below. Then run the cell to make sure the Cloud SDK uses the right project for all the commands in this notebook." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "GtIjHUhlJ7-W" + }, + "outputs": [], + "source": [ + "MY_PROJECT = \"YOUR PROJECT ID\"\n", + "MY_STAGING_BUCKET = \"gs://YOUR BUCKET\" # bucket should be in same region as ucaip" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "itgJdQPGJ7-W" + }, + "source": [ + "## Set up SDK workspace" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "o_wnT10RJ7-W" + }, + "outputs": [], + "source": [ + "import uuid\n", + "\n", + "import tensorflow as tf\n", + "from google.cloud import aiplatform\n", + "from tabulate import tabulate" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "O8XJZB3gR8eL" + }, + "source": [ + "## Initialize Vertex SDK for Python\n", + "\n", + "Initialize the *client* for Vertex AI." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Wrlk2B2nJ7-X" + }, + "outputs": [], + "source": [ + "aiplatform.init(project=MY_PROJECT, staging_bucket=MY_STAGING_BUCKET)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "rQMGw0XlPh-o" + }, + "source": [ + "## Create Training Script that saves Explainable model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "jNBmoLbhJ7-X" + }, + "outputs": [], + "source": [ + "%%writefile training_script.py\n", + "# Single, Mirror and Multi-Machine Distributed Training for CIFAR-10\n", + "\n", + "from explainable_ai_sdk.metadata.tf.v2 import SavedModelMetadataBuilder\n", + "\n", + "from tensorflow.python.client import device_lib\n", + "import tensorflow_datasets as tfds\n", + "import tensorflow as tf\n", + "\n", + "import numpy as np\n", + "import tempfile\n", + "import argparse\n", + "import sys\n", + "import os\n", + "\n", + "tfds.disable_progress_bar()\n", + "\n", + "parser = argparse.ArgumentParser()\n", + "parser.add_argument('--model-dir', dest='model_dir',\n", + " default=os.getenv('AIP_MODEL_DIR'), type=str, help='Model dir.')\n", + "parser.add_argument('--lr', dest='lr',\n", + " default=0.001, type=float,\n", + " help='Learning rate.')\n", + "parser.add_argument('--epochs', dest='epochs',\n", + " default=20, type=int,\n", + " help='Number of epochs.')\n", + "parser.add_argument('--steps', dest='steps',\n", + " default=100, type=int,\n", + " help='Number of steps per epoch.')\n", + "parser.add_argument('--distribute', dest='distribute', type=str, default='single',\n", + " help='distributed training strategy')\n", + "args = parser.parse_args()\n", + "\n", + "print('Python Version = {}'.format(sys.version))\n", + "print('TensorFlow Version = {}'.format(tf.__version__))\n", + "print('TF_CONFIG = {}'.format(os.environ.get('TF_CONFIG', 'Not found')))\n", + "\n", + "# Single Machine, single compute device\n", + "if args.distribute == 'single':\n", + " if tf.test.is_gpu_available():\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/gpu:0\")\n", + " else:\n", + " strategy = tf.distribute.OneDeviceStrategy(device=\"/cpu:0\")\n", + "# Single Machine, multiple compute device\n", + "elif args.distribute == 'mirror':\n", + " strategy = tf.distribute.MirroredStrategy()\n", + "# Multiple Machine, multiple compute device\n", + "elif args.distribute == 'multi':\n", + " strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "\n", + "# Multi-worker configuration\n", + "print('num_replicas_in_sync = {}'.format(strategy.num_replicas_in_sync))\n", + "\n", + "def make_dataset():\n", + " # Scaling Boston Housing data features\n", + " def scale(feature):\n", + " max = np.max(feature)\n", + " feature = (feature / max).astype(np.float)\n", + " return feature\n", + "\n", + " (x_train, y_train), (x_test, y_test) = tf.keras.datasets.boston_housing.load_data(\n", + " path=\"boston_housing.npz\", test_split=0.2, seed=113\n", + " )\n", + " for _ in range(13):\n", + " x_train[_] = scale(x_train[_])\n", + " x_test[_] = scale(x_test[_])\n", + " return (x_train, y_train), (x_test, y_test)\n", + "\n", + "# Build the Keras model\n", + "def build_and_compile_dnn_model():\n", + " model = tf.keras.Sequential([\n", + " tf.keras.layers.Dense(128, activation='relu', input_shape=(13,)),\n", + " tf.keras.layers.Dense(128, activation='relu'),\n", + " tf.keras.layers.Dense(1, activation='linear')\n", + " ])\n", + " model.compile(\n", + " loss='mse',\n", + " optimizer=tf.keras.optimizers.RMSprop(learning_rate=args.lr))\n", + " return model\n", + "\n", + "# Train the model\n", + "NUM_WORKERS = strategy.num_replicas_in_sync\n", + "# Here the batch size scales up by number of workers since\n", + "# `tf.data.Dataset.batch` expects the global batch size.\n", + "BATCH_SIZE = 16\n", + "GLOBAL_BATCH_SIZE = BATCH_SIZE * NUM_WORKERS\n", + "\n", + "with strategy.scope():\n", + " # Creation of dataset, and model building/compiling need to be within\n", + " # `strategy.scope()`.\n", + " model = build_and_compile_dnn_model()\n", + "\n", + "# Train the model\n", + "(x_train, y_train), (x_test, y_test) = make_dataset()\n", + "model.fit(x_train, y_train, epochs=args.epochs, batch_size=GLOBAL_BATCH_SIZE)\n", + "\n", + "tmpdir = tempfile.mkdtemp()\n", + "\n", + "model.save(tmpdir)\n", + "\n", + "# Save TF Model with Explainable metadata to GCS\n", + "builder = SavedModelMetadataBuilder(tmpdir)\n", + "builder.save_model_with_metadata(args.model_dir)\n" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "Yp2clkOJSDhR" + }, + "source": [ + "## Launch a Training Job and Create a Model on Vertex AI" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "FTtte5rGSGhc" + }, + "source": [ + "### Config a Training Job" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "OwmqEIKYJ7-Y" + }, + "outputs": [], + "source": [ + "job = aiplatform.CustomTrainingJob(\n", + " display_name=f\"temp-mbsdk-explainable-ai-custom-tabular-nb-{uuid.uuid4()}\",\n", + " script_path=\"training_script.py\",\n", + " container_uri=\"gcr.io/cloud-aiplatform/training/tf-gpu.2-1:latest\",\n", + " requirements=[\n", + " \"tensorflow_datasets\",\n", + " \"explainable-ai-sdk\",\n", + " ],\n", + " model_serving_container_image_uri=\"gcr.io/cloud-aiplatform/prediction/tf2-gpu.2-1:latest\",\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "FdqpORPhSLbx" + }, + "source": [ + "### Run the Training Job" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "oVTORjQpJ7-Y" + }, + "outputs": [], + "source": [ + "model = job.run(\n", + " model_display_name=\"temp-boston-housing-mbsdk-explainable-tabular-model\",\n", + " replica_count=1,\n", + " machine_type=\"n1-standard-4\",\n", + " accelerator_type=\"NVIDIA_TESLA_K80\",\n", + " accelerator_count=1,\n", + " args=[\"--epochs=50\", \"--distribute=single\"],\n", + ")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "uMnxLeMrJ7-Y" + }, + "outputs": [], + "source": [ + "# Get info about the Custom Job\n", + "print(\n", + " f\"Display Name:\\t{job.display_name}\\n\"\n", + " f\"Resource Name:\\t{job.resource_name}\\n\"\n", + " f\"Current State:\\t{job.state.name}\\n\"\n", + ")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "2Udmt7tpJ7-Y" + }, + "outputs": [], + "source": [ + "# Get path to saved model in GCS\n", + "output_dir = model._gca_resource.artifact_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "vKlHRCOPJ7-Z" + }, + "source": [ + "## Build the Explanation Metadata and Parameters" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "6z9zrDmTJ7-Z" + }, + "outputs": [], + "source": [ + "loaded = tf.keras.models.load_model(output_dir)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "jbp3HRc7J7-Z" + }, + "outputs": [], + "source": [ + "serving_input = list(\n", + " loaded.signatures[\"serving_default\"].structured_input_signature[1].keys()\n", + ")[0]\n", + "serving_output = list(loaded.signatures[\"serving_default\"].structured_outputs.keys())[0]\n", + "feature_names = [\n", + " \"crim\",\n", + " \"zn\",\n", + " \"indus\",\n", + " \"chas\",\n", + " \"nox\",\n", + " \"rm\",\n", + " \"age\",\n", + " \"dis\",\n", + " \"rad\",\n", + " \"tax\",\n", + " \"ptratio\",\n", + " \"b\",\n", + " \"lstat\",\n", + "]" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "xpYFtjRbJ7-Z" + }, + "outputs": [], + "source": [ + "explain_params = aiplatform.explain.ExplanationParameters(\n", + " {\"sampled_shapley_attribution\": {\"path_count\": 10}}\n", + ")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "AtLFH_FeJ7-a" + }, + "outputs": [], + "source": [ + "input_metadata = {\n", + " \"input_tensor_name\": serving_input,\n", + " \"encoding\": \"BAG_OF_FEATURES\",\n", + " \"modality\": \"numeric\",\n", + " \"index_feature_mapping\": feature_names,\n", + "}\n", + "output_metadata = {\"output_tensor_name\": serving_output}\n", + "\n", + "input_metadata = aiplatform.explain.ExplanationMetadata.InputMetadata(input_metadata)\n", + "output_metadata = aiplatform.explain.ExplanationMetadata.OutputMetadata(output_metadata)\n", + "\n", + "explain_metadata = aiplatform.explain.ExplanationMetadata(\n", + " inputs={\"features\": input_metadata}, outputs={\"medv\": output_metadata}\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "744xmYcKJ7-a" + }, + "source": [ + "## Deploy the model with model explanations enabled" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "IByGRREtJ7-a" + }, + "outputs": [], + "source": [ + "endpoint = model.deploy(\n", + " machine_type=\"n1-standard-4\",\n", + " accelerator_type=\"NVIDIA_TESLA_K80\",\n", + " accelerator_count=1,\n", + " explanation_metadata=explain_metadata,\n", + " explanation_parameters=explain_params,\n", + ")" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Gx0EJFHXxtqv" + }, + "outputs": [], + "source": [ + "print(f\"Endpoint resource name: {endpoint.resource_name}\")\n", + "print(\n", + " f\"\\nTo use this endpoint in the future:\\nendpoint = aiplatform.Endpoint('{endpoint.resource_name}')\"\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "E3K2TSrPJ7-a" + }, + "source": [ + "## Fetch test data to use" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "UYrdURLsJ7-b" + }, + "outputs": [], + "source": [ + "import numpy as np\n", + "from tensorflow.keras.datasets import boston_housing\n", + "\n", + "(_, _), (x_test, y_test) = boston_housing.load_data(\n", + " path=\"boston_housing.npz\", test_split=0.2, seed=113\n", + ")\n", + "\n", + "\n", + "def scale(feature):\n", + " max = np.max(feature)\n", + " feature = (feature / max).astype(np.float32)\n", + " return feature\n", + "\n", + "\n", + "for _ in range(13):\n", + " x_test[_] = scale(x_test[_])\n", + "x_test = x_test.astype(np.float32)\n", + "\n", + "print(x_test.shape, x_test.dtype, y_test.shape)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "s_fNdLGPJ7-b" + }, + "source": [ + "## Get predictions with explanations on our deployed tabular model" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "AhEjJQCsJ7-b" + }, + "outputs": [], + "source": [ + "response = endpoint.explain(\n", + " instances=[{\"dense_input\": s.tolist()} for s in [x_test[0]]]\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "8FBZDKVsJ7-b" + }, + "source": [ + "## Check out feature attributions" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "urngajW6J7-b" + }, + "outputs": [], + "source": [ + "test_data = x_test[0]\n", + "attributions = response.explanations[0].attributions[0].feature_attributions\n", + "\n", + "rows = []\n", + "for i, val in enumerate(feature_names):\n", + " rows.append([val, test_data[i], attributions[val][0]])\n", + "print(tabulate(rows, headers=[\"Feature name\", \"Feature value\", \"Attribution value\"]))" + ] + } + ], + "metadata": { + "colab": { + "collapsed_sections": [], + "name": "AI_Platform_(Unified)_SDK_Explainable_AI_Custom_Tabular.ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/sdk/SDK_Tabular_Custom_Model_Training_asynchronous.ipynb b/notebooks/community/sdk/SDK_Tabular_Custom_Model_Training_asynchronous.ipynb new file mode 100644 index 000000000..e744054a4 --- /dev/null +++ b/notebooks/community/sdk/SDK_Tabular_Custom_Model_Training_asynchronous.ipynb @@ -0,0 +1,357 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "ur8xi4C7S06n" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "1c955a52adac" + }, + "source": [ + "# Feedback or issues?\n", + "For any feedback or questions, please open an [issue](https://github.com/googleapis/python-aiplatform/issues)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "eHLV0D7Y5jtU" + }, + "source": [ + "# Vertex SDK for Python: Custom Tabular Training (asynchronous) Example\n", + "\n", + "To use this Colaboratory notebook, you copy the notebook to your own Google Drive and open it with Colaboratory (or Colab). You can run each step, or cell, and see its results. To run a cell, use Shift+Enter. Colab automatically displays the return value of the last line in each cell. For more information about running notebooks in Colab, see the [Colab welcome page](https://colab.research.google.com/notebooks/welcome.ipynb).\n", + "\n", + "This notebook demonstrate how to create a custom model based on a tabular dataset (asynchronously). It will require you provide a bucket where the dataset CSV will be stored.\n", + "\n", + "Note: you may incur charges for training, prediction, storage or usage of other GCP products in connection with testing this SDK." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "lld3eeJUs5yM" + }, + "source": [ + "# Install Vertex SDK for Python, Authenticate, and upload of a Dataset to your GCS bucket\n", + "\n", + "\n", + "After the SDK installation the kernel will be automatically restarted. You may see this error message `Your session crashed for an unknown reason` which is normal." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "sBfZtR4X1Dr_" + }, + "outputs": [], + "source": [ + "!pip3 uninstall -y google-cloud-aiplatform\n", + "!pip3 install google-cloud-aiplatform\n", + "import IPython\n", + "\n", + "app = IPython.Application.instance()\n", + "app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "c0SNmTBeD2nV" + }, + "source": [ + "### Enter your project and GCS bucket\n", + "\n", + "Enter your Project Id in the cell below. Then run the cell to make sure the Cloud SDK uses the right project for all the commands in this notebook." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "f5be5dce7259" + }, + "outputs": [], + "source": [ + "import sys\n", + "\n", + "if \"google.colab\" in sys.modules:\n", + " from google.colab import auth\n", + "\n", + " auth.authenticate_user()" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "kng9iKBwqcS5" + }, + "outputs": [], + "source": [ + "MY_PROJECT = \"YOUR PROJECT\"\n", + "MY_STAGING_BUCKET = \"gs://YOUR BUCKET\" # bucket should be in same region as Vertex AI" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "K-YsTXeQxXp7" + }, + "source": [ + "The dataset we are using is the Abalone Dataset. For more information about this dataset please visit: https://archive.ics.uci.edu/ml/datasets/abalone" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "V_T10yTTqcS_" + }, + "outputs": [], + "source": [ + "!wget https://storage.googleapis.com/download.tensorflow.org/data/abalone_train.csv\n", + "!gsutil cp abalone_train.csv {MY_STAGING_BUCKET}/data/\n", + "\n", + "gcs_csv_path = f\"{MY_STAGING_BUCKET}/data/abalone_train.csv\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "rk43VP_IqcTE" + }, + "source": [ + "# Initialize Vertex SDK for Python\n", + "\n", + "Initialize the *client* for Vertex AI" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "VCiC9gBWqcTF" + }, + "outputs": [], + "source": [ + "from google.cloud import aiplatform\n", + "\n", + "aiplatform.init(project=MY_PROJECT, staging_bucket=MY_STAGING_BUCKET)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "35QVNhACqcTJ" + }, + "source": [ + "# Create a Managed Tabular Dataset from CSV\n", + "\n", + "A Managed dataset can be used to create an AutoML model or a custom model. " + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "4OfCqaYRqcTJ" + }, + "outputs": [], + "source": [ + "ds = aiplatform.TabularDataset.create(\n", + " display_name=\"abalone\", gcs_source=[gcs_csv_path], sync=False\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "VcEOYYolqcTN" + }, + "source": [ + "# Write Training Script\n", + "- Write this cell as a file which will be used for custom training." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "OauJqJmJqcTO" + }, + "outputs": [], + "source": [ + "%%writefile training_script.py\n", + "\n", + "import pandas as pd\n", + "import os\n", + "import tensorflow as tf\n", + "from tensorflow import keras\n", + "from tensorflow.keras import layers\n", + "\n", + "\n", + "# uncomment and bump up replica_count for distributed training\n", + "# strategy = tf.distribute.experimental.MultiWorkerMirroredStrategy()\n", + "# tf.distribute.experimental_set_strategy(strategy)\n", + "\n", + "col_names = [\"Length\", \"Diameter\", \"Height\", \"Whole weight\", \"Shucked weight\", \"Viscera weight\", \"Shell weight\", \"Age\"]\n", + "target = \"Age\"\n", + "\n", + "def aip_data_to_dataframe(wild_card_path):\n", + " return pd.concat([pd.read_csv(fp.numpy().decode(), names=col_names)\n", + " for fp in tf.data.Dataset.list_files([wild_card_path])])\n", + "\n", + "def get_features_and_labels(df):\n", + " return df.drop(target, axis=1).values, df[target].values\n", + "\n", + "def data_prep(wild_card_path):\n", + " return get_features_and_labels(aip_data_to_dataframe(wild_card_path))\n", + "\n", + "\n", + "model = tf.keras.Sequential([layers.Dense(64), layers.Dense(1)])\n", + "model.compile(loss='mse', optimizer='adam')\n", + "\n", + "model.fit(*data_prep(os.environ[\"AIP_TRAINING_DATA_URI\"]),\n", + " epochs=10 ,\n", + " validation_data=data_prep(os.environ[\"AIP_VALIDATION_DATA_URI\"]))\n", + "print(model.evaluate(*data_prep(os.environ[\"AIP_TEST_DATA_URI\"])))\n", + "\n", + "# save as Vertex AI Managed model\n", + "tf.saved_model.save(model, os.environ[\"AIP_MODEL_DIR\"])" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "6-bBqipfqcTS" + }, + "source": [ + "# Launch a Training Job to Create a Model\n", + "\n", + "Once we have defined your training script, we will create a model." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "btb6d48lqcTT" + }, + "outputs": [], + "source": [ + "job = aiplatform.CustomTrainingJob(\n", + " display_name=\"train-abalone-dist-1-replica\",\n", + " script_path=\"training_script.py\",\n", + " container_uri=\"gcr.io/cloud-aiplatform/training/tf-cpu.2-2:latest\",\n", + " requirements=[\"gcsfs==0.7.1\"],\n", + " model_serving_container_image_uri=\"gcr.io/cloud-aiplatform/prediction/tf2-cpu.2-2:latest\",\n", + ")\n", + "model = job.run(ds, replica_count=1, model_display_name=\"abalone-model\", sync=False)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "5vhDsMJNqcTW" + }, + "source": [ + "# Deploy Your Model\n", + "\n", + "Deploy your model, then wait until the model FINISHES deployment before proceeding to prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "Y9GH72wWqcTX" + }, + "outputs": [], + "source": [ + "endpoint = model.deploy(machine_type=\"n1-standard-4\", sync=False)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "nIw1ifPuqcTb" + }, + "source": [ + "Wait for the deployment to complete" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "b75WevFnWsNP" + }, + "outputs": [], + "source": [ + "endpoint.wait()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "XIyKTohyPpSd" + }, + "source": [ + "# Predict on the Endpoint" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "3k6-rSZPqcTc" + }, + "outputs": [], + "source": [ + "prediction = endpoint.predict(\n", + " [\n", + " [0.435, 0.335, 0.11, 0.33399999999999996, 0.1355, 0.0775, 0.0965],\n", + " [0.585, 0.45, 0.125, 0.874, 0.3545, 0.2075, 0.225],\n", + " ]\n", + ")\n", + "prediction" + ] + } + ], + "metadata": { + "colab": { + "collapsed_sections": [], + "name": "AI_Platform_(Unified)_SDK_Tabular_Custom_Model_Training_(asynchronous).ipynb", + "toc_visible": true + }, + "kernelspec": { + "display_name": "Python 3", + "name": "python3" + } + }, + "nbformat": 4, + "nbformat_minor": 0 +} diff --git a/notebooks/community/sdk/sdk_automl_image_classification_batch.ipynb b/notebooks/community/sdk/sdk_automl_image_classification_batch.ipynb new file mode 100644 index 000000000..cbc4e6633 --- /dev/null +++ b/notebooks/community/sdk/sdk_automl_image_classification_batch.ipynb @@ -0,0 +1,1180 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "# Vertex SDK: AutoML training image classification model for batch prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex SDK to create image classification models and do batch prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:flowers,icn", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Flowers dataset](https://www.tensorflow.org/datasets/catalog/tf_flowers) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts the type of flower an image is from a class of five flowers: daisy, dandelion, rose, sunflower, or tulip." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,batch_prediction", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML image classification model from a Python script, and then do a batch prediction using the Vertex SDK. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Make a batch prediction.\n", + "\n", + "There is one key difference between using batch prediction and using online prediction:\n", + "\n", + "* Prediction Service: Does an on-demand prediction for the entire set of instances (i.e., one or more data items) and returns the results in real-time.\n", + "\n", + "* Batch Prediction Service: Does a queued (batch) prediction for the entire set of instances in the background and stores the results in a Cloud Storage bucket when ready." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex SDK." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "import sys\n", + "import os\n", + "\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = '--user'\n", + "else:\n", + " USER_FLAG = ''\n", + "\n", + "! pip3 install --upgrade google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" #@param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/ai-platform-unified/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "REGION = 'us-central1' #@param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you initialize the Vertex SDK for Python, you specify a Cloud Storage staging bucket. The staging bucket is where all the data associated with your dataset and model resources are retained across sessions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" #@param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "import google.cloud.aiplatform as aip" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "init_aip:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Initialize Vertex SDK\n", + "\n", + "Initialize the Vertex SDK for your project and corresponding bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "init_aip:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "aip.init(project=PROJECT_ID, staging_bucket=BUCKET_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML image classification model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_dataset", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "## Create a Dataset Resource\n", + "\n", + "First, you create an image Dataset resource for the Flowers dataset." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:image,u_dataset", + "repo": "snippets_common.ipynb" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for images has some requirements for your data:\n", + "\n", + "- Images must be stored in a Cloud Storage bucket.\n", + "- Each image file must be in an image format (PNG, JPEG, BMP, ...).\n", + "- There must be an index file stored in your Cloud Storage bucket that contains the path and label for each image.\n", + "- The index file must be either CSV or JSONL." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:icn,u_dataset,csv", + "repo": "snippets_common.ipynb" + }, + "source": [ + "#### CSV\n", + "\n", + "For image classification, the CSV index file has the requirements:\n", + "\n", + "- No heading.\n", + "- First column is the Cloud Storage path to the image.\n", + "- Second column is the label." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv", + "repo": "snippets_common.ipynb" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:flowers,csv,icn", + "repo": "snippets_common.ipynb" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = 'gs://cloud-samples-data/vision/automl_classification/flowers/all_data_v2.csv'" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:csv", + "repo": "snippets_common.ipynb" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Flowers dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:csv", + "repo": "snippets_common.ipynb" + }, + "outputs": [], + "source": [ + "if 'IMPORT_FILES' in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_dataset:image,icn", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Create the Dataset\n", + "\n", + "Next, create the `Dataset` resource using the `create()` method for the `ImageDataset` class, which takes the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the `Dataset` resource.\n", + "- `gcs_source`: A list of one or more dataset index file to import the data items into the `Dataset` resource.\n", + "- `import_schema_uri`: The data labeling schema for the data items.\n", + "\n", + "This operation may take several minutes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_dataset:image,icn", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "dataset = aip.ImageDataset.create(\n", + " display_name=\"Flowers\" + \"_\" + TIMESTAMP,\n", + " gcs_source=[IMPORT_FILE],\n", + " import_schema_uri=aip.schema.dataset.ioformat.image.single_label_classification,\n", + ")\n", + "\n", + "print(dataset.resource_name)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML image classification model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_automl_pipeline:image,icn", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Create and run training pipeline\n", + "\n", + "To train an AutoML image classification model, you perform two steps: 1) create a training pipeline, and 2) run the pipeline.\n", + "\n", + "#### Create training pipeline\n", + "\n", + "An AutoML training pipeline is created with the `AutoMLImageTrainingJob` class, with the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the `TrainingJob` resource.\n", + "- `prediction_type`: The type task to train the model for.\n", + " - `classification`: An image classification model.\n", + " - `object_detection`: An image object detection model.\n", + "- `multi_label`: If a classification task, whether single (`False`) or multi-labeled (`True`).\n", + "- `model_type`: The type of model for deployment.\n", + " - `CLOUD`: Deployment on Google Cloud\n", + " - `CLOUD_HIGH_ACCURACY_1`: Optimized for accuracy over latency for deployment on Google Cloud.\n", + " - `CLOUD_LOW_LATENCY_`: Optimized for latency over accuracy for deployment on Google Cloud.\n", + " - `MOBILE_TF_VERSATILE_1`: Deployment on an edge device.\n", + " - `MOBILE_TF_HIGH_ACCURACY_1`:Optimized for accuracy over latency for deployment on an edge device.\n", + " - `MOBILE_TF_LOW_LATENCY_1`: Optimized for latency over accuracy for deployment on an edge device.\n", + "- `base_model`: (optional) Transfer learning from existing `Model` resource -- supported for image classification only.\n", + "\n", + "The instantiated object is the DAG for the training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_automl_pipeline:image,icn", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "dag = aip.AutoMLImageTrainingJob(\n", + " display_name=\"flowers_\" + TIMESTAMP,\n", + " prediction_type=\"classification\",\n", + " multi_label=False,\n", + " model_type=\"CLOUD\",\n", + " base_model=None,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "run_automl_pipeline:image", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "#### Run the training pipeline\n", + "\n", + "Next, you run the DAG to start the training job by invoking the method `run()`, with the following parameters:\n", + "\n", + "- `dataset`: The `Dataset` resource to train the model.\n", + "- `model_display_name`: The human readable name for the trained model.\n", + "- `training_fraction_split`: The percentage of the dataset to use for training.\n", + "- `validation_fraction_split`: The percentage of the dataset to use for validation.\n", + "- `test_fraction_split`: The percentage of the dataset to use for test (holdout data).\n", + "- `budget_milli_node_hours`: (optional) Maximum training time specified in unit of millihours (1000 = hour).\n", + "- `disable_early_stopping`: If `True`, training maybe completed before using the entire budget if the service believes it cannot further improve on the model objective measurements.\n", + "\n", + "The `run` method when completed returns the `Model` resource.\n", + "\n", + "The execution of the training pipeline will take upto 20 minutes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "run_automl_pipeline:image", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "model = dag.run(\n", + " dataset=dataset,\n", + " model_display_name=\"flowers_\" + TIMESTAMP,\n", + " training_fraction_split=0.8,\n", + " validation_fraction_split=0.1,\n", + " test_fraction_split=0.1,\n", + " budget_milli_node_hours=8000,\n", + " disable_early_stopping=False\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy:batch_prediction", + "repo": "snippets_common.ipynb" + }, + "source": [ + "## Model deployment for batch prediction\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created for batch prediction. This differs from deploying a `Model` resource for online prediction.\n", + "\n", + "For online prediction, you:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource.\n", + "\n", + "3. Make online prediction requests to the `Endpoint` resource.\n", + "\n", + "For batch-prediction, you:\n", + "\n", + "1. Create a batch prediction job.\n", + "\n", + "2. The job service will provision resources for the batch prediction request.\n", + "\n", + "3. The results of the batch prediction request are returned to the caller.\n", + "\n", + "4. The job service will unprovision the resoures for the batch prediction request." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction", + "repo": "snippets_common.ipynb" + }, + "source": [ + "## Make a batch prediction request\n", + "\n", + "Now do a batch prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_items:batch_prediction", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "### Get test item(s)\n", + "\n", + "Now do a batch prediction to your Vertex model. You will use arbitrary examples out of the dataset as a test items. Don't be concerned that the examples were likely used in training the model -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_items:automl,icn,csv", + "repo": "snippets_automl.ipynb" + }, + "outputs": [], + "source": [ + "test_items = !gsutil cat $IMPORT_FILE | head -n2\n", + "if len(str(test_items[0]).split(',')) == 3:\n", + " _, test_item_1, test_label_1 = str(test_items[0]).split(',')\n", + " _, test_item_2, test_label_2 = str(test_items[1]).split(',')\n", + "else:\n", + " test_item_1, test_label_1 = str(test_items[0]).split(',')\n", + " test_item_2, test_label_2 = str(test_items[1]).split(',')\n", + "\n", + "print(test_item_1, test_label_1)\n", + "print(test_item_2, test_label_2)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "copy_test_items:batch_prediction", + "repo": "snippets_common.ipynb" + }, + "source": [ + "### Copy test item(s)\n", + "\n", + "For the batch prediction, you will copy the test items over to your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copy_test_items:batch_prediction", + "repo": "snippets_common.ipynb" + }, + "outputs": [], + "source": [ + "file_1 = test_item_1.split('/')[-1]\n", + "file_2 = test_item_2.split('/')[-1]\n", + "\n", + "! gsutil cp $test_item_1 $BUCKET_NAME/$file_1\n", + "! gsutil cp $test_item_2 $BUCKET_NAME/$file_2\n", + "\n", + "test_item_1 = BUCKET_NAME + \"/\" + file_1\n", + "test_item_2 = BUCKET_NAME + \"/\" + file_2" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:automl,image", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "Now make a batch input file, which you will store in your local Cloud Storage bucket. The batch input file can be either CSV or JSONL. You will use JSONL in this tutorial. For JSONL file, you make one dictionary entry per line for each data item (instance). The dictionary contains the key/value pairs:\n", + "\n", + "- `content`: The Cloud Storage path to the image.\n", + "- `mime_type`: The content type. In our example, it is an `jpeg` file.\n", + "\n", + "For example:\n", + "\n", + " {'content': '[your-bucket]/file1.jpg', 'mime_type': 'jpeg'}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_file:automl,image", + "repo": "snippets_automl.ipynb" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "import json\n", + "\n", + "gcs_input_uri = BUCKET_NAME + '/test.jsonl'\n", + "with tf.io.gfile.GFile(gcs_input_uri, 'w') as f:\n", + " data = {\"content\": test_item_1, \"mime_type\": \"image/jpeg\"}\n", + " f.write(json.dumps(data) + '\\n')\n", + " data = {\"content\": test_item_2, \"mime_type\": \"image/jpeg\"}\n", + " f.write(json.dumps(data) + '\\n')\n", + "\n", + "print(gcs_input_uri)\n", + "! gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_request:mbsdk,automl", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Make the batch prediction request\n", + "\n", + "Now that your `Model` resource is trained, you can make a batch prediction by invoking the `batch_request()` method, with the following parameters:\n", + "\n", + "- `job_display_name`: The human readable name for the batch prediction job.\n", + "- `gcs_source`: A list of one or more batch request input files.\n", + "- `gcs_destination_prefix`: The Cloud Storage location for storing the batch prediction resuls.\n", + "- `sync`: If set to `True`, the call will block while waiting for the asynchronous batch job to complete." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_request:mbsdk,automl", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "batch_predict_job = model.batch_predict(\n", + " job_display_name=\"$(DATASET_ALIAS)_\" + TIMESTAMP,\n", + " gcs_source=gcs_input_uri,\n", + " gcs_destination_prefix=BUCKET_NAME,\n", + " sync=False\n", + ")\n", + "\n", + "print(batch_predict_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_batch_job:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Wait for completion of batch prediction job\n", + "\n", + "Next, wait for the batch job to complete." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_batch_job:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "batch_predict_job.wait()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_batch_prediction:mbsdk,icn", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Get the predictions\n", + "\n", + "Next, get the results from the completed batch prediction job.\n", + "\n", + "The results are written to the Cloud Storage output bucket you specified in the batch prediction request. You call the method `iter_outputs()` to get a list of each Cloud Storage file generated with the results. Each file contains one or more prediction requests in a JSON format:\n", + "\n", + "- `content`: The prediction request.\n", + "- `prediction`: The prediction response.\n", + " - `ids`: The internal assigned unique identifiers for each prediction request.\n", + " - `displayNames`: The class names for each class label.\n", + " - `confidences`: The predicted confidence, between 0 and 1, per class label." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_batch_prediction:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "bp_iter_outputs = batch_predict_job.iter_outputs()\n", + "\n", + "prediction_results = list()\n", + "for blob in bp_iter_outputs:\n", + " if blob.name.split(\"/\")[-1].startswith(\"prediction\"):\n", + " prediction_results.append(blob.name)\n", + "\n", + "tags = list()\n", + "for prediction_result in prediction_results:\n", + " gfile_name = f\"gs://{bp_iter_outputs.bucket.name}/{prediction_result}\"\n", + " with tf.io.gfile.GFile(name=gfile_name, mode=\"r\") as gfile:\n", + " for line in gfile.readlines():\n", + " line = json.loads(line)\n", + " print(line)\n", + " break" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "\n", + "# Delete the dataset using the Vertex dataset object\n", + "try:\n", + " if delete_dataset and 'dataset' in globals():\n", + " dataset.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex model object\n", + "try:\n", + " if delete_model and 'model' in globals():\n", + " model.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex endpoint object\n", + "try:\n", + " if delete_endpoint and 'model' in globals():\n", + " endpoint.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch prediction job using the Vertex batch prediction object\n", + "try:\n", + " if delete_batchjob and 'model' in globals():\n", + " batch_predict_job.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and 'BUCKET_NAME' in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.7.6" + }, + "vars": { + "AIP": "AI Platform", + "AUTOML": "AutoML", + "AUTOML_URL": "https://cloud.google.com/vertex-ai/docs/start/automl-users", + "BATCH_OR_ONLINE": "batch", + "BQ": "BigQuery", + "COLAB": "Colab", + "DATASET_ALIAS": "flowers", + "DATASET_NAME": "Flowers", + "DATA_TYPE": "image", + "DOCKER": "Docker", + "EDGE": "Edge", + "EMAIL": "cdpe@gmail.com", + "GAPIC": "client library", + "GAPICP": "SDK", + "GCP": "Google Cloud", + "GCS": "Cloud Storage", + "GNOTEBOOK": "Google Cloud Notebook", + "HEIGHT": "128", + "IMPORT_FORMAT": "csv", + "MILLIHOURS": "8000", + "MODEL_TYPE": "image classification", + "NOTEBOOK": "sdk_automl_image_classification_batch.ipynb", + "REGION": "us-central1", + "REPO": "ai-platform-unified/notebooks/community/gapic/custom", + "TENSORFLOW": "TensorFlow", + "TFLite": "TFLite", + "TFServing": "TF Serving", + "TFV": "2-1", + "TITLE": "AutoML training image classification model for batch prediction", + "TRAINING": "training", + "TRAINING_TIME": "20", + "TRAINING_WAIT": "60", + "WIDTH": "128", + "YEAR": "2021", + "null": "null", + "uCAIP": "Vertex" + } + }, + "nbformat": 4, + "nbformat_minor": 4 +} diff --git a/notebooks/community/sdk/sdk_automl_image_classification_online.ipynb b/notebooks/community/sdk/sdk_automl_image_classification_online.ipynb new file mode 100644 index 000000000..934c50189 --- /dev/null +++ b/notebooks/community/sdk/sdk_automl_image_classification_online.ipynb @@ -0,0 +1,1069 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "# Copyright 2020 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "# Vertex SDK: AutoML training image classification model for online prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex SDK to create image classification models and do online prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:flowers,icn", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Flowers dataset](https://www.tensorflow.org/datasets/catalog/tf_flowers) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts the type of flower an image is from a class of five flowers: daisy, dandelion, rose, sunflower, or tulip." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,online_prediction", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML image classification model and deploy for online prediction from a Python script using the Vertex SDK. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model`." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex SDK." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "import sys\n", + "import os\n", + "\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = '--user'\n", + "else:\n", + " USER_FLAG = ''\n", + "\n", + "! pip3 install --upgrade google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" #@param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/ai-platform-unified/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "REGION = 'us-central1' #@param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you initialize the Vertex SDK for Python, you specify a Cloud Storage staging bucket. The staging bucket is where all the data associated with your dataset and model resources are retained across sessions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" #@param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "import google.cloud.aiplatform as aip" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "init_aip:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Initialize Vertex SDK\n", + "\n", + "Initialize the Vertex SDK for your project and corresponding bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "init_aip:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "aip.init(project=PROJECT_ID, staging_bucket=BUCKET_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML image classification model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_dataset", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "## Create a Dataset Resource\n", + "\n", + "First, you create an image Dataset resource for the Flowers dataset." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:image,u_dataset", + "repo": "snippets_common.ipynb" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for images has some requirements for your data:\n", + "\n", + "- Images must be stored in a Cloud Storage bucket.\n", + "- Each image file must be in an image format (PNG, JPEG, BMP, ...).\n", + "- There must be an index file stored in your Cloud Storage bucket that contains the path and label for each image.\n", + "- The index file must be either CSV or JSONL." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:icn,u_dataset,csv", + "repo": "snippets_common.ipynb" + }, + "source": [ + "#### CSV\n", + "\n", + "For image classification, the CSV index file has the requirements:\n", + "\n", + "- No heading.\n", + "- First column is the Cloud Storage path to the image.\n", + "- Second column is the label." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv", + "repo": "snippets_common.ipynb" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:flowers,csv,icn", + "repo": "snippets_common.ipynb" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = 'gs://cloud-samples-data/vision/automl_classification/flowers/all_data_v2.csv'" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:csv", + "repo": "snippets_common.ipynb" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Flowers dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:csv", + "repo": "snippets_common.ipynb" + }, + "outputs": [], + "source": [ + "if 'IMPORT_FILES' in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_dataset:image,icn", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Create the Dataset\n", + "\n", + "Next, create the `Dataset` resource using the `create()` method for the `ImageDataset` class, which takes the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the `Dataset` resource.\n", + "- `gcs_source`: A list of one or more dataset index file to import the data items into the `Dataset` resource.\n", + "- `import_schema_uri`: The data labeling schema for the data items.\n", + "\n", + "This operation may take several minutes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_dataset:image,icn", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "dataset = aip.ImageDataset.create(\n", + " display_name=\"Flowers\" + \"_\" + TIMESTAMP,\n", + " gcs_source=[IMPORT_FILE],\n", + " import_schema_uri=aip.schema.dataset.ioformat.image.single_label_classification,\n", + ")\n", + "\n", + "print(dataset.resource_name)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML image classification model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_automl_pipeline:image,icn", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Create and run training pipeline\n", + "\n", + "To train an AutoML image classification model, you perform two steps: 1) create a training pipeline, and 2) run the pipeline.\n", + "\n", + "#### Create training pipeline\n", + "\n", + "An AutoML training pipeline is created with the `AutoMLImageTrainingJob` class, with the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the `TrainingJob` resource.\n", + "- `prediction_type`: The type task to train the model for.\n", + " - `classification`: An image classification model.\n", + " - `object_detection`: An image object detection model.\n", + "- `multi_label`: If a classification task, whether single (`False`) or multi-labeled (`True`).\n", + "- `model_type`: The type of model for deployment.\n", + " - `CLOUD`: Deployment on Google Cloud\n", + " - `CLOUD_HIGH_ACCURACY_1`: Optimized for accuracy over latency for deployment on Google Cloud.\n", + " - `CLOUD_LOW_LATENCY_`: Optimized for latency over accuracy for deployment on Google Cloud.\n", + " - `MOBILE_TF_VERSATILE_1`: Deployment on an edge device.\n", + " - `MOBILE_TF_HIGH_ACCURACY_1`:Optimized for accuracy over latency for deployment on an edge device.\n", + " - `MOBILE_TF_LOW_LATENCY_1`: Optimized for latency over accuracy for deployment on an edge device.\n", + "- `base_model`: (optional) Transfer learning from existing `Model` resource -- supported for image classification only.\n", + "\n", + "The instantiated object is the DAG for the training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_automl_pipeline:image,icn", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "dag = aip.AutoMLImageTrainingJob(\n", + " display_name=\"flowers_\" + TIMESTAMP,\n", + " prediction_type=\"classification\",\n", + " multi_label=False,\n", + " model_type=\"CLOUD\",\n", + " base_model=None,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "run_automl_pipeline:image", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "#### Run the training pipeline\n", + "\n", + "Next, you run the DAG to start the training job by invoking the method `run()`, with the following parameters:\n", + "\n", + "- `dataset`: The `Dataset` resource to train the model.\n", + "- `model_display_name`: The human readable name for the trained model.\n", + "- `training_fraction_split`: The percentage of the dataset to use for training.\n", + "- `validation_fraction_split`: The percentage of the dataset to use for validation.\n", + "- `test_fraction_split`: The percentage of the dataset to use for test (holdout data).\n", + "- `budget_milli_node_hours`: (optional) Maximum training time specified in unit of millihours (1000 = hour).\n", + "- `disable_early_stopping`: If `True`, training maybe completed before using the entire budget if the service believes it cannot further improve on the model objective measurements.\n", + "\n", + "The `run` method when completed returns the `Model` resource.\n", + "\n", + "The execution of the training pipeline will take upto 20 minutes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "run_automl_pipeline:image", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "model = dag.run(\n", + " dataset=dataset,\n", + " model_display_name=\"flowers_\" + TIMESTAMP,\n", + " training_fraction_split=0.8,\n", + " validation_fraction_split=0.1,\n", + " test_fraction_split=0.1,\n", + " budget_milli_node_hours=8000,\n", + " disable_early_stopping=False\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "## Deploy the model\n", + "\n", + "Next, deploy your `Model` resource to an `Endpoint` resource for online prediction. To deploy the `Model` resource, you invoke the `deploy()` method. This call will create an `Endpoint` resource automatically.\n", + "\n", + "The method returns the created `Endpoint` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "endpoint = model.deploy()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction", + "repo": "snippets_common.ipynb" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_item", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "### Get test item\n", + "\n", + "You will use an arbitrary example out of the dataset as a test item. Don't be concerned that the example was likely used in training the model -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:automl,icn,csv", + "repo": "snippets_automl.ipynb" + }, + "outputs": [], + "source": [ + "test_item = !gsutil cat $IMPORT_FILE | head -n1\n", + "if len(str(test_item[0]).split(',')) == 3:\n", + " _, test_item, test_label = str(test_item[0]).split(',')\n", + "else:\n", + " test_item, test_label = str(test_item[0]).split(',')\n", + "\n", + "print(test_item, test_label)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "predict_request:mbsdk,image,icn", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Make the prediction\n", + "\n", + "Now that your `Model` resource is deployed to an `Endpoint` resource, one can do online predictions by sending prediction requests to the `Endpoint` resource.\n", + "\n", + "#### Request\n", + "\n", + "Since in this example your test item is in a Cloud Storage bucket, you open and read the contents of the image using `tf.io.gfile.Gfile()`. To pass the test data to the prediction service, you encode the bytes into base64 -- which makes the content safe from modification while transmitting binary data over the network.\n", + "\n", + "The format of each instance is:\n", + "\n", + " { 'content': { 'b64': [base64_encoded_bytes] } }\n", + "\n", + "Since the `predict()` method can take multiple items (instances), send your single test item as a list of one test item.\n", + "\n", + "#### Response\n", + "\n", + "The response from the `predict()` call is a Python dictionary with the following entries:\n", + "\n", + "- `ids`: The internal assigned unique identifiers for each prediction request.\n", + "- `displayNames`: The class names for each class label.\n", + "- `confidences`: The predicted confidence, between 0 and 1, per class label.\n", + "- `deployed_model_id`: The Vertex identifier for the deployed `Model` resource which did the predictions." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "predict_request:mbsdk,image,icn", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "import base64\n", + "\n", + "import tensorflow as tf\n", + "\n", + "with tf.io.gfile.GFile(test_item, \"rb\") as f:\n", + " content = f.read()\n", + "\n", + "# The format of each instance should conform to the deployed model's prediction input schema.\n", + "instances = [{\"content\": base64.b64encode(content).decode(\"utf-8\")}]\n", + "\n", + "prediction = endpoint.predict(instances=instances)\n", + "\n", + "print(prediction)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "## Undeploy the model\n", + "\n", + "When you are done doing predictions, you undeploy the `Model` resource from the `Endpoint` resouce. This deprovisions all compute resources and ends billing for the deployed model." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "endpoint.undeploy_all()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "\n", + "# Delete the dataset using the Vertex dataset object\n", + "try:\n", + " if delete_dataset and 'dataset' in globals():\n", + " dataset.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex model object\n", + "try:\n", + " if delete_model and 'model' in globals():\n", + " model.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex endpoint object\n", + "try:\n", + " if delete_endpoint and 'model' in globals():\n", + " endpoint.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch prediction job using the Vertex batch prediction object\n", + "try:\n", + " if delete_batchjob and 'model' in globals():\n", + " batch_predict_job.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and 'BUCKET_NAME' in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.7.6" + }, + "vars": { + "AIP": "AI Platform", + "AUTOML": "AutoML", + "AUTOML_URL": "https://cloud.google.com/vertex-ai/docs/start/automl-users", + "BATCH_OR_ONLINE": "online", + "BQ": "BigQuery", + "COLAB": "Colab", + "DATASET_ALIAS": "flowers", + "DATASET_NAME": "Flowers", + "DATA_TYPE": "image", + "DOCKER": "Docker", + "EDGE": "Edge", + "EMAIL": "cdpe@gmail.com", + "GAPIC": "client library", + "GAPICP": "SDK", + "GCP": "Google Cloud", + "GCS": "Cloud Storage", + "GNOTEBOOK": "Google Cloud Notebook", + "HEIGHT": "128", + "IMPORT_FORMAT": "csv", + "MILLIHOURS": "8000", + "MODEL_TYPE": "image classification", + "NOTEBOOK": "sdk_automl_image_classification_online.ipynb", + "REGION": "us-central1", + "REPO": "ai-platform-unified/notebooks/community/gapic/custom", + "TENSORFLOW": "TensorFlow", + "TFLite": "TFLite", + "TFServing": "TF Serving", + "TFV": "2-1", + "TITLE": "AutoML training image classification model for online prediction", + "TRAINING": "training", + "TRAINING_TIME": "20", + "TRAINING_WAIT": "60", + "WIDTH": "128", + "YEAR": "2020", + "null": "null", + "uCAIP": "Vertex" + } + }, + "nbformat": 4, + "nbformat_minor": 4 +} diff --git a/notebooks/community/sdk/sdk_automl_image_object_detection_batch.ipynb b/notebooks/community/sdk/sdk_automl_image_object_detection_batch.ipynb new file mode 100644 index 000000000..8b04322b6 --- /dev/null +++ b/notebooks/community/sdk/sdk_automl_image_object_detection_batch.ipynb @@ -0,0 +1,1189 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "# Vertex SDK: AutoML training image object detection model for batch prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex SDK to create image object detection models and do batch prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:salads,iod", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the Salads category of the [OpenImages dataset](https://www.tensorflow.org/datasets/catalog/open_images_v4) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). This dataset does not require any feature engineering. The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts the bounding box locations and corresponding type of salad items in an image from a class of five items: salad, seafood, tomato, baked goods, or cheese." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,batch_prediction", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML image object detection model from a Python script, and then do a batch prediction using the Vertex SDK. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Make a batch prediction.\n", + "\n", + "There is one key difference between using batch prediction and using online prediction:\n", + "\n", + "* Prediction Service: Does an on-demand prediction for the entire set of instances (i.e., one or more data items) and returns the results in real-time.\n", + "\n", + "* Batch Prediction Service: Does a queued (batch) prediction for the entire set of instances in the background and stores the results in a Cloud Storage bucket when ready." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex SDK." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "import sys\n", + "import os\n", + "\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = '--user'\n", + "else:\n", + " USER_FLAG = ''\n", + "\n", + "! pip3 install --upgrade google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" #@param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/ai-platform-unified/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "REGION = 'us-central1' #@param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you initialize the Vertex SDK for Python, you specify a Cloud Storage staging bucket. The staging bucket is where all the data associated with your dataset and model resources are retained across sessions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" #@param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "import google.cloud.aiplatform as aip" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "init_aip:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Initialize Vertex SDK\n", + "\n", + "Initialize the Vertex SDK for your project and corresponding bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "init_aip:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "aip.init(project=PROJECT_ID, staging_bucket=BUCKET_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML image object detection model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_dataset", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "## Create a Dataset Resource\n", + "\n", + "First, you create an image Dataset resource for the Salads dataset." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:image,u_dataset", + "repo": "snippets_common.ipynb" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for images has some requirements for your data:\n", + "\n", + "- Images must be stored in a Cloud Storage bucket.\n", + "- Each image file must be in an image format (PNG, JPEG, BMP, ...).\n", + "- There must be an index file stored in your Cloud Storage bucket that contains the path and label for each image.\n", + "- The index file must be either CSV or JSONL." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:iod,u_dataset,csv", + "repo": "snippets_common.ipynb" + }, + "source": [ + "#### CSV\n", + "\n", + "For image object detection, the CSV index file has the requirements:\n", + "\n", + "- No heading.\n", + "- First column is the Cloud Storage path to the image.\n", + "- Second column is the label.\n", + "- Third/Fourth columns are the upper left corner of bounding box. Coordinates are normalized, between 0 and 1.\n", + "- Fifth/Sixth/Seventh columns are not used and should be 0.\n", + "- Eighth/Ninth columns are the lower right corner of the bounding box." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv", + "repo": "snippets_common.ipynb" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:salads,csv,iod", + "repo": "snippets_common.ipynb" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = 'gs://cloud-samples-data/vision/salads.csv'" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:csv", + "repo": "snippets_common.ipynb" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Salads dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:csv", + "repo": "snippets_common.ipynb" + }, + "outputs": [], + "source": [ + "if 'IMPORT_FILES' in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_dataset:image,iod", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Create the Dataset\n", + "\n", + "Next, create the `Dataset` resource using the `create()` method for the `ImageDataset` class, which takes the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the `Dataset` resource.\n", + "- `gcs_source`: A list of one or more dataset index file to import the data items into the `Dataset` resource.\n", + "- `import_schema_uri`: The data labeling schema for the data items.\n", + "\n", + "This operation may take several minutes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_dataset:image,iod", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "dataset = aip.ImageDataset.create(\n", + " display_name=\"Salads\" + \"_\" + TIMESTAMP,\n", + " gcs_source=[IMPORT_FILE],\n", + " import_schema_uri=aip.schema.dataset.ioformat.image.bounding_box,\n", + ")\n", + "\n", + "print(dataset.resource_name)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML image object detection model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_automl_pipeline:image,iod", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Create and run training pipeline\n", + "\n", + "To train an AutoML image object detection model, you perform two steps: 1) create a training pipeline, and 2) run the pipeline.\n", + "\n", + "#### Create training pipeline\n", + "\n", + "An AutoML training pipeline is created with the `AutoMLImageTrainingJob` class, with the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the `TrainingJob` resource.\n", + "- `prediction_type`: The type task to train the model for.\n", + " - `classification`: An image classification model.\n", + " - `object_detection`: An image object detection model.\n", + "- `multi_label`: If a classification task, whether single (`False`) or multi-labeled (`True`).\n", + "- `model_type`: The type of model for deployment.\n", + " - `CLOUD`: Deployment on Google Cloud\n", + " - `CLOUD_HIGH_ACCURACY_1`: Optimized for accuracy over latency for deployment on Google Cloud.\n", + " - `CLOUD_LOW_LATENCY_`: Optimized for latency over accuracy for deployment on Google Cloud.\n", + " - `MOBILE_TF_VERSATILE_1`: Deployment on an edge device.\n", + " - `MOBILE_TF_HIGH_ACCURACY_1`:Optimized for accuracy over latency for deployment on an edge device.\n", + " - `MOBILE_TF_LOW_LATENCY_1`: Optimized for latency over accuracy for deployment on an edge device.\n", + "- `base_model`: (optional) Transfer learning from existing `Model` resource -- supported for image classification only.\n", + "\n", + "The instantiated object is the DAG for the training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_automl_pipeline:image,iod", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "dag = aip.AutoMLImageTrainingJob(\n", + " display_name=\"salads_\" + TIMESTAMP,\n", + " prediction_type=\"object_detection\",\n", + " model_type=\"CLOUD\",\n", + " base_model=None,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "run_automl_pipeline:image", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "#### Run the training pipeline\n", + "\n", + "Next, you run the DAG to start the training job by invoking the method `run()`, with the following parameters:\n", + "\n", + "- `dataset`: The `Dataset` resource to train the model.\n", + "- `model_display_name`: The human readable name for the trained model.\n", + "- `training_fraction_split`: The percentage of the dataset to use for training.\n", + "- `validation_fraction_split`: The percentage of the dataset to use for validation.\n", + "- `test_fraction_split`: The percentage of the dataset to use for test (holdout data).\n", + "- `budget_milli_node_hours`: (optional) Maximum training time specified in unit of millihours (1000 = hour).\n", + "- `disable_early_stopping`: If `True`, training maybe completed before using the entire budget if the service believes it cannot further improve on the model objective measurements.\n", + "\n", + "The `run` method when completed returns the `Model` resource.\n", + "\n", + "The execution of the training pipeline will take upto 20 minutes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "run_automl_pipeline:image", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "model = dag.run(\n", + " dataset=dataset,\n", + " model_display_name=\"salads_\" + TIMESTAMP,\n", + " training_fraction_split=0.8,\n", + " validation_fraction_split=0.1,\n", + " test_fraction_split=0.1,\n", + " budget_milli_node_hours=20000,\n", + " disable_early_stopping=False\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy:batch_prediction", + "repo": "snippets_common.ipynb" + }, + "source": [ + "## Model deployment for batch prediction\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created for batch prediction. This differs from deploying a `Model` resource for online prediction.\n", + "\n", + "For online prediction, you:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource.\n", + "\n", + "3. Make online prediction requests to the `Endpoint` resource.\n", + "\n", + "For batch-prediction, you:\n", + "\n", + "1. Create a batch prediction job.\n", + "\n", + "2. The job service will provision resources for the batch prediction request.\n", + "\n", + "3. The results of the batch prediction request are returned to the caller.\n", + "\n", + "4. The job service will unprovision the resoures for the batch prediction request." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction", + "repo": "snippets_common.ipynb" + }, + "source": [ + "## Make a batch prediction request\n", + "\n", + "Now do a batch prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_items:batch_prediction", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "### Get test item(s)\n", + "\n", + "Now do a batch prediction to your Vertex model. You will use arbitrary examples out of the dataset as a test items. Don't be concerned that the examples were likely used in training the model -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_items:automl,iod,csv", + "repo": "snippets_automl.ipynb" + }, + "outputs": [], + "source": [ + "test_items = !gsutil cat $IMPORT_FILE | head -n2\n", + "cols_1 = str(test_items[0]).split(',')\n", + "cols_2 = str(test_items[1]).split(',')\n", + "if len(cols_1) == 11:\n", + " test_item_1 = str(cols_1[1])\n", + " test_label_1 = str(cols_1[2])\n", + " test_item_2 = str(cols_2[1])\n", + " test_label_2 = str(cols_2[2])\n", + "else:\n", + " test_item_1 = str(cols_1[0])\n", + " test_label_1 = str(cols_1[1])\n", + " test_item_2 = str(cols_2[0])\n", + " test_label_2 = str(cols_2[1])\n", + "\n", + "print(test_item_1, test_label_1)\n", + "print(test_item_2, test_label_2)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "copy_test_items:batch_prediction", + "repo": "snippets_common.ipynb" + }, + "source": [ + "### Copy test item(s)\n", + "\n", + "For the batch prediction, you will copy the test items over to your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copy_test_items:batch_prediction", + "repo": "snippets_common.ipynb" + }, + "outputs": [], + "source": [ + "file_1 = test_item_1.split('/')[-1]\n", + "file_2 = test_item_2.split('/')[-1]\n", + "\n", + "! gsutil cp $test_item_1 $BUCKET_NAME/$file_1\n", + "! gsutil cp $test_item_2 $BUCKET_NAME/$file_2\n", + "\n", + "test_item_1 = BUCKET_NAME + \"/\" + file_1\n", + "test_item_2 = BUCKET_NAME + \"/\" + file_2" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:automl,image", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "Now make a batch input file, which you will store in your local Cloud Storage bucket. The batch input file can be either CSV or JSONL. You will use JSONL in this tutorial. For JSONL file, you make one dictionary entry per line for each data item (instance). The dictionary contains the key/value pairs:\n", + "\n", + "- `content`: The Cloud Storage path to the image.\n", + "- `mime_type`: The content type. In our example, it is an `jpeg` file.\n", + "\n", + "For example:\n", + "\n", + " {'content': '[your-bucket]/file1.jpg', 'mime_type': 'jpeg'}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_file:automl,image", + "repo": "snippets_automl.ipynb" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "import json\n", + "\n", + "gcs_input_uri = BUCKET_NAME + '/test.jsonl'\n", + "with tf.io.gfile.GFile(gcs_input_uri, 'w') as f:\n", + " data = {\"content\": test_item_1, \"mime_type\": \"image/jpeg\"}\n", + " f.write(json.dumps(data) + '\\n')\n", + " data = {\"content\": test_item_2, \"mime_type\": \"image/jpeg\"}\n", + " f.write(json.dumps(data) + '\\n')\n", + "\n", + "print(gcs_input_uri)\n", + "! gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_request:mbsdk,automl", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Make the batch prediction request\n", + "\n", + "Now that your `Model` resource is trained, you can make a batch prediction by invoking the `batch_request()` method, with the following parameters:\n", + "\n", + "- `job_display_name`: The human readable name for the batch prediction job.\n", + "- `gcs_source`: A list of one or more batch request input files.\n", + "- `gcs_destination_prefix`: The Cloud Storage location for storing the batch prediction resuls.\n", + "- `sync`: If set to `True`, the call will block while waiting for the asynchronous batch job to complete." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_request:mbsdk,automl", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "batch_predict_job = model.batch_predict(\n", + " job_display_name=\"$(DATASET_ALIAS)_\" + TIMESTAMP,\n", + " gcs_source=gcs_input_uri,\n", + " gcs_destination_prefix=BUCKET_NAME,\n", + " sync=False\n", + ")\n", + "\n", + "print(batch_predict_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_batch_job:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Wait for completion of batch prediction job\n", + "\n", + "Next, wait for the batch job to complete." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_batch_job:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "batch_predict_job.wait()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_batch_prediction:mbsdk,iod", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Get the predictions\n", + "\n", + "Next, get the results from the completed batch prediction job.\n", + "\n", + "The results are written to the Cloud Storage output bucket you specified in the batch prediction request. You call the method `iter_outputs()` to get a list of each Cloud Storage file generated with the results. Each file contains one or more prediction requests in a JSON format:\n", + "\n", + "- `content`: The prediction request.\n", + "- `prediction`: The prediction response.\n", + " - `ids`: The internal assigned unique identifiers for each prediction request.\n", + " - `displayNames`: The class names for each class label.\n", + " - `confidences`: The predicted confidence of each object, between 0 and 1, per class label.\n", + " - `bboxes`: The bounding box for each object" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_batch_prediction:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "bp_iter_outputs = batch_predict_job.iter_outputs()\n", + "\n", + "prediction_results = list()\n", + "for blob in bp_iter_outputs:\n", + " if blob.name.split(\"/\")[-1].startswith(\"prediction\"):\n", + " prediction_results.append(blob.name)\n", + "\n", + "tags = list()\n", + "for prediction_result in prediction_results:\n", + " gfile_name = f\"gs://{bp_iter_outputs.bucket.name}/{prediction_result}\"\n", + " with tf.io.gfile.GFile(name=gfile_name, mode=\"r\") as gfile:\n", + " for line in gfile.readlines():\n", + " line = json.loads(line)\n", + " print(line)\n", + " break" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "\n", + "# Delete the dataset using the Vertex dataset object\n", + "try:\n", + " if delete_dataset and 'dataset' in globals():\n", + " dataset.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex model object\n", + "try:\n", + " if delete_model and 'model' in globals():\n", + " model.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex endpoint object\n", + "try:\n", + " if delete_endpoint and 'model' in globals():\n", + " endpoint.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch prediction job using the Vertex batch prediction object\n", + "try:\n", + " if delete_batchjob and 'model' in globals():\n", + " batch_predict_job.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and 'BUCKET_NAME' in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.7.6" + }, + "vars": { + "AIP": "AI Platform", + "AUTOML": "AutoML", + "AUTOML_URL": "https://cloud.google.com/vertex-ai/docs/start/automl-users", + "BATCH_OR_ONLINE": "batch", + "BQ": "BigQuery", + "COLAB": "Colab", + "DATASET_ALIAS": "salads", + "DATASET_NAME": "Salads", + "DATA_TYPE": "image", + "DOCKER": "Docker", + "EDGE": "Edge", + "EMAIL": "cdpe@gmail.com", + "GAPIC": "client library", + "GAPICP": "SDK", + "GCP": "Google Cloud", + "GCS": "Cloud Storage", + "GNOTEBOOK": "Google Cloud Notebook", + "HEIGHT": "128", + "IMPORT_FORMAT": "csv", + "MILLIHOURS": "20000", + "MODEL_TYPE": "image object detection", + "NOTEBOOK": "sdk_automl_image_object_detection_batch.ipynb", + "REGION": "us-central1", + "REPO": "ai-platform-unified/notebooks/community/gapic/custom", + "TENSORFLOW": "TensorFlow", + "TFLite": "TFLite", + "TFServing": "TF Serving", + "TFV": "2-1", + "TITLE": "AutoML training image object detection model for batch prediction", + "TRAINING": "training", + "TRAINING_TIME": "20", + "TRAINING_WAIT": "60", + "WIDTH": "128", + "YEAR": "2021", + "null": "null", + "uCAIP": "Vertex" + } + }, + "nbformat": 4, + "nbformat_minor": 4 +} diff --git a/notebooks/community/sdk/sdk_automl_image_object_detection_online.ipynb b/notebooks/community/sdk/sdk_automl_image_object_detection_online.ipynb new file mode 100644 index 000000000..569416255 --- /dev/null +++ b/notebooks/community/sdk/sdk_automl_image_object_detection_online.ipynb @@ -0,0 +1,1075 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "# Vertex SDK: AutoML training image object detection model for online prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex SDK to create image object detection models and do online prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:salads,iod", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the Salads category of the [OpenImages dataset](https://www.tensorflow.org/datasets/catalog/open_images_v4) from [TensorFlow Datasets](https://www.tensorflow.org/datasets/catalog/overview). This dataset does not require any feature engineering. The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket. The trained model predicts the bounding box locations and corresponding type of salad items in an image from a class of five items: salad, seafood, tomato, baked goods, or cheese." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,online_prediction", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML image object detection model and deploy for online prediction from a Python script using the Vertex SDK. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model`." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex SDK." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "import sys\n", + "import os\n", + "\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = '--user'\n", + "else:\n", + " USER_FLAG = ''\n", + "\n", + "! pip3 install --upgrade google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" #@param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/ai-platform-unified/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "REGION = 'us-central1' #@param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you initialize the Vertex SDK for Python, you specify a Cloud Storage staging bucket. The staging bucket is where all the data associated with your dataset and model resources are retained across sessions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" #@param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "import google.cloud.aiplatform as aip" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "init_aip:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Initialize Vertex SDK\n", + "\n", + "Initialize the Vertex SDK for your project and corresponding bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "init_aip:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "aip.init(project=PROJECT_ID, staging_bucket=BUCKET_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML image object detection model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_dataset", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "## Create a Dataset Resource\n", + "\n", + "First, you create an image Dataset resource for the Salads dataset." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:image,u_dataset", + "repo": "snippets_common.ipynb" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for images has some requirements for your data:\n", + "\n", + "- Images must be stored in a Cloud Storage bucket.\n", + "- Each image file must be in an image format (PNG, JPEG, BMP, ...).\n", + "- There must be an index file stored in your Cloud Storage bucket that contains the path and label for each image.\n", + "- The index file must be either CSV or JSONL." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:iod,u_dataset,csv", + "repo": "snippets_common.ipynb" + }, + "source": [ + "#### CSV\n", + "\n", + "For image object detection, the CSV index file has the requirements:\n", + "\n", + "- No heading.\n", + "- First column is the Cloud Storage path to the image.\n", + "- Second column is the label.\n", + "- Third/Fourth columns are the upper left corner of bounding box. Coordinates are normalized, between 0 and 1.\n", + "- Fifth/Sixth/Seventh columns are not used and should be 0.\n", + "- Eighth/Ninth columns are the lower right corner of the bounding box." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv", + "repo": "snippets_common.ipynb" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:salads,csv,iod", + "repo": "snippets_common.ipynb" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = 'gs://cloud-samples-data/vision/salads.csv'" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:csv", + "repo": "snippets_common.ipynb" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Salads dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:csv", + "repo": "snippets_common.ipynb" + }, + "outputs": [], + "source": [ + "if 'IMPORT_FILES' in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_dataset:image,iod", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Create the Dataset\n", + "\n", + "Next, create the `Dataset` resource using the `create()` method for the `ImageDataset` class, which takes the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the `Dataset` resource.\n", + "- `gcs_source`: A list of one or more dataset index file to import the data items into the `Dataset` resource.\n", + "- `import_schema_uri`: The data labeling schema for the data items.\n", + "\n", + "This operation may take several minutes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_dataset:image,iod", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "dataset = aip.ImageDataset.create(\n", + " display_name=\"Salads\" + \"_\" + TIMESTAMP,\n", + " gcs_source=[IMPORT_FILE],\n", + " import_schema_uri=aip.schema.dataset.ioformat.image.bounding_box,\n", + ")\n", + "\n", + "print(dataset.resource_name)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML image object detection model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_automl_pipeline:image,iod", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Create and run training pipeline\n", + "\n", + "To train an AutoML image object detection model, you perform two steps: 1) create a training pipeline, and 2) run the pipeline.\n", + "\n", + "#### Create training pipeline\n", + "\n", + "An AutoML training pipeline is created with the `AutoMLImageTrainingJob` class, with the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the `TrainingJob` resource.\n", + "- `prediction_type`: The type task to train the model for.\n", + " - `classification`: An image classification model.\n", + " - `object_detection`: An image object detection model.\n", + "- `multi_label`: If a classification task, whether single (`False`) or multi-labeled (`True`).\n", + "- `model_type`: The type of model for deployment.\n", + " - `CLOUD`: Deployment on Google Cloud\n", + " - `CLOUD_HIGH_ACCURACY_1`: Optimized for accuracy over latency for deployment on Google Cloud.\n", + " - `CLOUD_LOW_LATENCY_`: Optimized for latency over accuracy for deployment on Google Cloud.\n", + " - `MOBILE_TF_VERSATILE_1`: Deployment on an edge device.\n", + " - `MOBILE_TF_HIGH_ACCURACY_1`:Optimized for accuracy over latency for deployment on an edge device.\n", + " - `MOBILE_TF_LOW_LATENCY_1`: Optimized for latency over accuracy for deployment on an edge device.\n", + "- `base_model`: (optional) Transfer learning from existing `Model` resource -- supported for image classification only.\n", + "\n", + "The instantiated object is the DAG for the training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_automl_pipeline:image,iod", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "dag = aip.AutoMLImageTrainingJob(\n", + " display_name=\"salads_\" + TIMESTAMP,\n", + " prediction_type=\"object_detection\",\n", + " model_type=\"CLOUD\",\n", + " base_model=None,\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "run_automl_pipeline:image", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "#### Run the training pipeline\n", + "\n", + "Next, you run the DAG to start the training job by invoking the method `run()`, with the following parameters:\n", + "\n", + "- `dataset`: The `Dataset` resource to train the model.\n", + "- `model_display_name`: The human readable name for the trained model.\n", + "- `training_fraction_split`: The percentage of the dataset to use for training.\n", + "- `validation_fraction_split`: The percentage of the dataset to use for validation.\n", + "- `test_fraction_split`: The percentage of the dataset to use for test (holdout data).\n", + "- `budget_milli_node_hours`: (optional) Maximum training time specified in unit of millihours (1000 = hour).\n", + "- `disable_early_stopping`: If `True`, training maybe completed before using the entire budget if the service believes it cannot further improve on the model objective measurements.\n", + "\n", + "The `run` method when completed returns the `Model` resource.\n", + "\n", + "The execution of the training pipeline will take upto 20 minutes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "run_automl_pipeline:image", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "model = dag.run(\n", + " dataset=dataset,\n", + " model_display_name=\"salads_\" + TIMESTAMP,\n", + " training_fraction_split=0.8,\n", + " validation_fraction_split=0.1,\n", + " test_fraction_split=0.1,\n", + " budget_milli_node_hours=20000,\n", + " disable_early_stopping=False\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "## Deploy the model\n", + "\n", + "Next, deploy your `Model` resource to an `Endpoint` resource for online prediction. To deploy the `Model` resource, you invoke the `deploy()` method. This call will create an `Endpoint` resource automatically.\n", + "\n", + "The method returns the created `Endpoint` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "endpoint = model.deploy()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction", + "repo": "snippets_common.ipynb" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_item", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "### Get test item\n", + "\n", + "You will use an arbitrary example out of the dataset as a test item. Don't be concerned that the example was likely used in training the model -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:automl,iod,csv", + "repo": "snippets_automl.ipynb" + }, + "outputs": [], + "source": [ + "test_items = !gsutil cat $IMPORT_FILE | head -n1\n", + "cols = str(test_items[0]).split(',')\n", + "if len(cols) == 11:\n", + " test_item = str(cols[1])\n", + " test_label = str(cols[2])\n", + "else:\n", + " test_item = str(cols[0])\n", + " test_label = str(cols[1])\n", + "\n", + "print(test_item, test_label)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "predict_request:mbsdk,image,iod", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Make the prediction\n", + "\n", + "Now that your `Model` resource is deployed to an `Endpoint` resource, one can do online predictions by sending prediction requests to the `Endpoint` resource.\n", + "\n", + "#### Request\n", + "\n", + "Since in this example your test item is in a Cloud Storage bucket, you open and read the contents of the image using `tf.io.gfile.Gfile()`. To pass the test data to the prediction service, you encode the bytes into base64 -- which makes the content safe from modification while transmitting binary data over the network.\n", + "\n", + "The format of each instance is:\n", + "\n", + " { 'content': { 'b64': [base64_encoded_bytes] } }\n", + "\n", + "Since the `predict()` method can take multiple items (instances), send your single test item as a list of one test item.\n", + "\n", + "#### Response\n", + "\n", + "The response from the `predict()` call is a Python dictionary with the following entries:\n", + "\n", + "- `ids`: The internal assigned unique identifiers for each prediction request.\n", + "- `displayNames`: The class names for each class label.\n", + "- `confidences`: The predicted confidence of each detected object, between 0 and 1, per class label.\n", + "- `bboxes`: The bounding box of each detected object.\n", + "- `deployed_model_id`: The Vertex identifier for the deployed `Model` resource which did the predictions." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "predict_request:mbsdk,image,iod", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "import base64\n", + "\n", + "import tensorflow as tf\n", + "\n", + "with tf.io.gfile.GFile(test_item, \"rb\") as f:\n", + " content = f.read()\n", + "\n", + "# The format of each instance should conform to the deployed model's prediction input schema.\n", + "instances = [{\"content\": base64.b64encode(content).decode(\"utf-8\")}]\n", + "\n", + "prediction = endpoint.predict(instances=instances)\n", + "\n", + "print(prediction)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "## Undeploy the model\n", + "\n", + "When you are done doing predictions, you undeploy the `Model` resource from the `Endpoint` resouce. This deprovisions all compute resources and ends billing for the deployed model." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "endpoint.undeploy_all()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "\n", + "# Delete the dataset using the Vertex dataset object\n", + "try:\n", + " if delete_dataset and 'dataset' in globals():\n", + " dataset.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex model object\n", + "try:\n", + " if delete_model and 'model' in globals():\n", + " model.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex endpoint object\n", + "try:\n", + " if delete_endpoint and 'model' in globals():\n", + " endpoint.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch prediction job using the Vertex batch prediction object\n", + "try:\n", + " if delete_batchjob and 'model' in globals():\n", + " batch_predict_job.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and 'BUCKET_NAME' in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.7.6" + }, + "vars": { + "AIP": "AI Platform", + "AUTOML": "AutoML", + "AUTOML_URL": "https://cloud.google.com/vertex-ai/docs/start/automl-users", + "BATCH_OR_ONLINE": "online", + "BQ": "BigQuery", + "COLAB": "Colab", + "DATASET_ALIAS": "salads", + "DATASET_NAME": "Salads", + "DATA_TYPE": "image", + "DOCKER": "Docker", + "EDGE": "Edge", + "EMAIL": "cdpe@gmail.com", + "GAPIC": "client library", + "GAPICP": "SDK", + "GCP": "Google Cloud", + "GCS": "Cloud Storage", + "GNOTEBOOK": "Google Cloud Notebook", + "HEIGHT": "128", + "IMPORT_FORMAT": "csv", + "MILLIHOURS": "20000", + "MODEL_TYPE": "image object detection", + "NOTEBOOK": "sdk_automl_image_object_detection_online.ipynb", + "REGION": "us-central1", + "REPO": "ai-platform-unified/notebooks/community/gapic/custom", + "TENSORFLOW": "TensorFlow", + "TFLite": "TFLite", + "TFServing": "TF Serving", + "TFV": "2-1", + "TITLE": "AutoML training image object detection model for online prediction", + "TRAINING": "training", + "TRAINING_TIME": "20", + "TRAINING_WAIT": "60", + "WIDTH": "128", + "YEAR": "2021", + "null": "null", + "uCAIP": "Vertex" + } + }, + "nbformat": 4, + "nbformat_minor": 4 +} diff --git a/notebooks/community/sdk/sdk_automl_tabular_binary_classification_batch.ipynb b/notebooks/community/sdk/sdk_automl_tabular_binary_classification_batch.ipynb new file mode 100644 index 000000000..25a99a54b --- /dev/null +++ b/notebooks/community/sdk/sdk_automl_tabular_binary_classification_batch.ipynb @@ -0,0 +1,1132 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "# Vertex SDK: AutoML training tabular binary classification model for batch prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex SDK to create tabular binary classification models and do batch prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:bank,lbn", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Bank Marketing](gs://cloud-ml-tables-data/bank-marketing.csv). This dataset does not require any feature engineering. The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,batch_prediction", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML tabular binary classification model from a Python script, and then do a batch prediction using the Vertex SDK. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Make a batch prediction.\n", + "\n", + "There is one key difference between using batch prediction and using online prediction:\n", + "\n", + "* Prediction Service: Does an on-demand prediction for the entire set of instances (i.e., one or more data items) and returns the results in real-time.\n", + "\n", + "* Batch Prediction Service: Does a queued (batch) prediction for the entire set of instances in the background and stores the results in a Cloud Storage bucket when ready." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex SDK." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "import sys\n", + "import os\n", + "\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = '--user'\n", + "else:\n", + " USER_FLAG = ''\n", + "\n", + "! pip3 install --upgrade google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" #@param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/ai-platform-unified/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "REGION = 'us-central1' #@param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you initialize the Vertex SDK for Python, you specify a Cloud Storage staging bucket. The staging bucket is where all the data associated with your dataset and model resources are retained across sessions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" #@param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "import google.cloud.aiplatform as aip" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "init_aip:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Initialize Vertex SDK\n", + "\n", + "Initialize the Vertex SDK for your project and corresponding bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "init_aip:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "aip.init(project=PROJECT_ID, staging_bucket=BUCKET_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML tabular binary classification model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_dataset", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "## Create a Dataset Resource\n", + "\n", + "First, you create an tabular Dataset resource for the Bank Marketing dataset." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:tabular,u_dataset", + "repo": "snippets_common.ipynb" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for tabular has a couple of requirements for your tabular data.\n", + "\n", + "- Must be in a CSV file or a BigQuery query." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:lbn,u_dataset,csv", + "repo": "snippets_common.ipynb" + }, + "source": [ + "#### CSV\n", + "\n", + "For tabular binary classification, the CSV file has a few requirements:\n", + "\n", + "- The first row must be the heading -- note how this is different from Vision, Video and Language where the requirement is no heading.\n", + "- All but one column are features.\n", + "- One column is the label, which you will specify when you subsequently create the training pipeline." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv", + "repo": "snippets_common.ipynb" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:bank,csv,lbn", + "repo": "snippets_common.ipynb" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = 'gs://cloud-ml-tables-data/bank-marketing.csv'" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:tabular", + "repo": "snippets_common.ipynb" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Bank Marketing dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows.\n", + "\n", + "You also need for training to know the heading name of the label column, which is save as `label_column`. For this dataset, it is the last column in the CSV file." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:tabular", + "repo": "snippets_common.ipynb" + }, + "outputs": [], + "source": [ + "count = ! gsutil cat $IMPORT_FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $IMPORT_FILE | head\n", + "\n", + "heading = ! gsutil cat $IMPORT_FILE | head -n1\n", + "label_column = str(heading).split(',')[-1].split(\"'\")[0]\n", + "print(\"Label Column Name\", label_column)\n", + "if label_column is None:\n", + " raise Exception(\"label column missing\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_dataset:tabular,lbn", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Create the Dataset\n", + "\n", + "Next, create the `Dataset` resource using the `create()` method for the `TabularDataset` class, which takes the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the `Dataset` resource.\n", + "- `gcs_source`: A list of one or more dataset index file to import the data items into the `Dataset` resource.\n", + "\n", + "This operation may take several minutes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_dataset:tabular,lbn", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "dataset = aip.TabularDataset.create(\n", + " display_name=\"Bank Marketing\" + \"_\" + TIMESTAMP,\n", + " gcs_source=[IMPORT_FILE]\n", + ")\n", + "\n", + "print(dataset.resource_name)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML tabular binary classification model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_automl_pipeline:tabular,lbn", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Create and run training pipeline\n", + "\n", + "To train an AutoML tabular binary classification model, you perform two steps: 1) create a training pipeline, and 2) run the pipeline.\n", + "\n", + "#### Create training pipeline\n", + "\n", + "An AutoML training pipeline is created with the `AutoMLTabularTrainingJob` class, with the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the `TrainingJob` resource.\n", + "- `optimization_prediction_type`: The type task to train the model for.\n", + " - `classification`: A tabuar classification model.\n", + " - `regression`: A tabular regression model.\n", + " - `forecasting`: A tabular forecasting model.\n", + "- `column_transformations`: (Optional): Transformations to apply to the input columns\n", + "- `optimization_objective`: The optimization objective to minimize or maximize.\n", + " - `minimize-log-loss`" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_automl_pipeline:tabular,lbn", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "dag = aip.AutoMLTabularTrainingJob(\n", + " display_name=\"bank_\" + TIMESTAMP,\n", + " optimization_prediction_type=\"classification\",\n", + " optimization_objective=\"minimize-log-loss\"\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "run_automl_pipeline:tabular", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "#### Run the training pipeline\n", + "\n", + "Next, you run the DAG to start the training job by invoking the method `run()`, with the following parameters:\n", + "\n", + "- `dataset`: The `Dataset` resource to train the model.\n", + "- `model_display_name`: The human readable name for the trained model.\n", + "- `target_column`: The name of the column to train as the label.\n", + "- `training_fraction_split`: The percentage of the dataset to use for training.\n", + "- `validation_fraction_split`: The percentage of the dataset to use for validation.\n", + "- `test_fraction_split`: The percentage of the dataset to use for test (holdout data).\n", + "- `budget_milli_node_hours`: (optional) Maximum training time specified in unit of millihours (1000 = hour).\n", + "- `disable_early_stopping`: If `True`, training maybe completed before using the entire budget if the service believes it cannot further improve on the model objective measurements.\n", + "\n", + "The `run` method when completed returns the `Model` resource.\n", + "\n", + "The execution of the training pipeline will take upto 20 minutes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "run_automl_pipeline:tabular", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "\n", + "model = dag.run(\n", + " dataset=dataset,\n", + " target_column=label_column,\n", + " model_display_name=\"bank_\" + TIMESTAMP,\n", + " training_fraction_split=0.6,\n", + " validation_fraction_split=0.2,\n", + " test_fraction_split=0.2,\n", + " budget_milli_node_hours=1000,\n", + " disable_early_stopping=False\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy:batch_prediction", + "repo": "snippets_common.ipynb" + }, + "source": [ + "## Model deployment for batch prediction\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created for batch prediction. This differs from deploying a `Model` resource for online prediction.\n", + "\n", + "For online prediction, you:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource.\n", + "\n", + "3. Make online prediction requests to the `Endpoint` resource.\n", + "\n", + "For batch-prediction, you:\n", + "\n", + "1. Create a batch prediction job.\n", + "\n", + "2. The job service will provision resources for the batch prediction request.\n", + "\n", + "3. The results of the batch prediction request are returned to the caller.\n", + "\n", + "4. The job service will unprovision the resoures for the batch prediction request." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction", + "repo": "snippets_common.ipynb" + }, + "source": [ + "## Make a batch prediction request\n", + "\n", + "Now do a batch prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_test_items:automl,batch_prediction", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "### Make test items\n", + "\n", + "You will use synthetic data as a test data items. Don't be concerned that we are using synthetic data -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_test_items:automl,tabular,bank,csv", + "repo": "snippets_automl.ipynb" + }, + "outputs": [], + "source": [ + "HEADING = \"Age,Job,MaritalStatus,Education,Default,Balance,Housing,Loan,Contact,Day,Month,Duration,Campaign,PDays,Previous,POutcome,Deposit\"\n", + "INSTANCE_1 = \"58,managment,married,teritary,no,2143,yes,no,unknown,5,may,261,1,-1,0, unknown\"\n", + "INSTANCE_2 = \"44,technician,single,secondary,no,39,yes,no,unknown,5,may,151,1,-1,0,unknown\"" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:automl,tabular", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "Now make a batch input file, which you will store in your local Cloud Storage bucket. Unlike image, video and text, the batch input file for tabular is only supported for CSV. For CSV file, you make:\n", + "\n", + "- The first line is the heading with the feature (fields) heading names.\n", + "- Each remaining line is a separate prediction request with the corresponding feature values.\n", + "\n", + "For example:\n", + "\n", + " \"feature_1\", \"feature_2\". ...\n", + " value_1, value_2, ..." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_file:automl,tabular", + "repo": "snippets_automl.ipynb" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "\n", + "gcs_input_uri = BUCKET_NAME + '/test.csv'\n", + "with tf.io.gfile.GFile(gcs_input_uri, 'w') as f:\n", + " f.write(HEADING + '\\n')\n", + " f.write(str(INSTANCE_1) + '\\n')\n", + " f.write(str(INSTANCE_2) + '\\n')\n", + "\n", + "print(gcs_input_uri)\n", + "! gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_request:mbsdk,automl", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Make the batch prediction request\n", + "\n", + "Now that your `Model` resource is trained, you can make a batch prediction by invoking the `batch_request()` method, with the following parameters:\n", + "\n", + "- `job_display_name`: The human readable name for the batch prediction job.\n", + "- `gcs_source`: A list of one or more batch request input files.\n", + "- `gcs_destination_prefix`: The Cloud Storage location for storing the batch prediction resuls.\n", + "- `sync`: If set to `True`, the call will block while waiting for the asynchronous batch job to complete." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_request:mbsdk,automl", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "batch_predict_job = model.batch_predict(\n", + " job_display_name=\"$(DATASET_ALIAS)_\" + TIMESTAMP,\n", + " gcs_source=gcs_input_uri,\n", + " gcs_destination_prefix=BUCKET_NAME,\n", + " sync=False\n", + ")\n", + "\n", + "print(batch_predict_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_batch_job:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Wait for completion of batch prediction job\n", + "\n", + "Next, wait for the batch job to complete." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_batch_job:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "batch_predict_job.wait()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_batch_prediction:mbsdk,lbn", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Get the predictions\n", + "\n", + "Next, get the results from the completed batch prediction job.\n", + "\n", + "The results are written to the Cloud Storage output bucket you specified in the batch prediction request. You call the method `iter_outputs()` to get a list of each Cloud Storage file generated with the results. Each file contains one or more prediction requests in a JSON format:\n", + "\n", + "- `content`: The prediction request.\n", + "- `prediction`: The prediction response.\n", + " - `ids`: The internal assigned unique identifiers for each prediction request.\n", + " - `displayNames`: The class names for each class label.\n", + " - `confidences`: The predicted confidence, between 0 and 1, per class label." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_batch_prediction:mbsdk,csv", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "bp_iter_outputs = batch_predict_job.iter_outputs()\n", + "\n", + "prediction_results = list()\n", + "for blob in bp_iter_outputs:\n", + " if blob.name.split(\"/\")[-1].startswith(\"prediction\"):\n", + " prediction_results.append(blob.name)\n", + "\n", + "tags = list()\n", + "for prediction_result in prediction_results:\n", + " gfile_name = f\"gs://{bp_iter_outputs.bucket.name}/{prediction_result}\"\n", + " with tf.io.gfile.GFile(name=gfile_name, mode=\"r\") as gfile:\n", + " for line in gfile.readlines():\n", + " print(line)\n", + " break" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "\n", + "# Delete the dataset using the Vertex dataset object\n", + "try:\n", + " if delete_dataset and 'dataset' in globals():\n", + " dataset.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex model object\n", + "try:\n", + " if delete_model and 'model' in globals():\n", + " model.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex endpoint object\n", + "try:\n", + " if delete_endpoint and 'model' in globals():\n", + " endpoint.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch prediction job using the Vertex batch prediction object\n", + "try:\n", + " if delete_batchjob and 'model' in globals():\n", + " batch_predict_job.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and 'BUCKET_NAME' in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.7.6" + }, + "vars": { + "AIP": "AI Platform", + "AUTOML": "AutoML", + "AUTOML_URL": "https://cloud.google.com/vertex-ai/docs/start/automl-users", + "BATCH_OR_ONLINE": "batch", + "BQ": "BigQuery", + "COLAB": "Colab", + "DATASET_ALIAS": "bank", + "DATASET_NAME": "Bank Marketing", + "DATA_TYPE": "tabular", + "DOCKER": "Docker", + "EDGE": "Edge", + "EMAIL": "cdpe@gmail.com", + "GAPIC": "client library", + "GAPICP": "SDK", + "GCP": "Google Cloud", + "GCS": "Cloud Storage", + "GNOTEBOOK": "Google Cloud Notebook", + "HEIGHT": "128", + "IMPORT_FORMAT": "csv", + "MILLIHOURS": "1000", + "MODEL_TYPE": "tabular binary classification", + "NOTEBOOK": "sdk_automl_tabular_binary_classification_batch.ipynb", + "REGION": "us-central1", + "REPO": "ai-platform-unified/notebooks/community/gapic/custom", + "TENSORFLOW": "TensorFlow", + "TFLite": "TFLite", + "TFServing": "TF Serving", + "TFV": "2-1", + "TITLE": "AutoML training tabular binary classification model for batch prediction", + "TRAINING": "training", + "TRAINING_TIME": "20", + "TRAINING_WAIT": "60", + "WIDTH": "128", + "YEAR": "2021", + "null": "null", + "uCAIP": "Vertex" + } + }, + "nbformat": 4, + "nbformat_minor": 4 +} diff --git a/notebooks/community/sdk/sdk_automl_tabular_binary_classification_online.ipynb b/notebooks/community/sdk/sdk_automl_tabular_binary_classification_online.ipynb new file mode 100644 index 000000000..c58c29c58 --- /dev/null +++ b/notebooks/community/sdk/sdk_automl_tabular_binary_classification_online.ipynb @@ -0,0 +1,1048 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "# Vertex SDK: AutoML training tabular binary classification model for online prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex SDK to create tabular binary classification models and do online prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:bank,lbn", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Bank Marketing](gs://cloud-ml-tables-data/bank-marketing.csv). This dataset does not require any feature engineering. The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,online_prediction", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML tabular binary classification model and deploy for online prediction from a Python script using the Vertex SDK. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model`." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex SDK." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "import sys\n", + "import os\n", + "\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = '--user'\n", + "else:\n", + " USER_FLAG = ''\n", + "\n", + "! pip3 install --upgrade google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" #@param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/ai-platform-unified/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "REGION = 'us-central1' #@param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you initialize the Vertex SDK for Python, you specify a Cloud Storage staging bucket. The staging bucket is where all the data associated with your dataset and model resources are retained across sessions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" #@param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "import google.cloud.aiplatform as aip" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "init_aip:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Initialize Vertex SDK\n", + "\n", + "Initialize the Vertex SDK for your project and corresponding bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "init_aip:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "aip.init(project=PROJECT_ID, staging_bucket=BUCKET_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML tabular binary classification model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_dataset", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "## Create a Dataset Resource\n", + "\n", + "First, you create an tabular Dataset resource for the Bank Marketing dataset." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:tabular,u_dataset", + "repo": "snippets_common.ipynb" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for tabular has a couple of requirements for your tabular data.\n", + "\n", + "- Must be in a CSV file or a BigQuery query." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:lbn,u_dataset,csv", + "repo": "snippets_common.ipynb" + }, + "source": [ + "#### CSV\n", + "\n", + "For tabular binary classification, the CSV file has a few requirements:\n", + "\n", + "- The first row must be the heading -- note how this is different from Vision, Video and Language where the requirement is no heading.\n", + "- All but one column are features.\n", + "- One column is the label, which you will specify when you subsequently create the training pipeline." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv", + "repo": "snippets_common.ipynb" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:bank,csv,lbn", + "repo": "snippets_common.ipynb" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = 'gs://cloud-ml-tables-data/bank-marketing.csv'" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:tabular", + "repo": "snippets_common.ipynb" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Bank Marketing dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows.\n", + "\n", + "You also need for training to know the heading name of the label column, which is save as `label_column`. For this dataset, it is the last column in the CSV file." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:tabular", + "repo": "snippets_common.ipynb" + }, + "outputs": [], + "source": [ + "count = ! gsutil cat $IMPORT_FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $IMPORT_FILE | head\n", + "\n", + "heading = ! gsutil cat $IMPORT_FILE | head -n1\n", + "label_column = str(heading).split(',')[-1].split(\"'\")[0]\n", + "print(\"Label Column Name\", label_column)\n", + "if label_column is None:\n", + " raise Exception(\"label column missing\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_dataset:tabular,lbn", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Create the Dataset\n", + "\n", + "Next, create the `Dataset` resource using the `create()` method for the `TabularDataset` class, which takes the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the `Dataset` resource.\n", + "- `gcs_source`: A list of one or more dataset index file to import the data items into the `Dataset` resource.\n", + "\n", + "This operation may take several minutes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_dataset:tabular,lbn", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "dataset = aip.TabularDataset.create(\n", + " display_name=\"Bank Marketing\" + \"_\" + TIMESTAMP,\n", + " gcs_source=[IMPORT_FILE]\n", + ")\n", + "\n", + "print(dataset.resource_name)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML tabular binary classification model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_automl_pipeline:tabular,lbn", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Create and run training pipeline\n", + "\n", + "To train an AutoML tabular binary classification model, you perform two steps: 1) create a training pipeline, and 2) run the pipeline.\n", + "\n", + "#### Create training pipeline\n", + "\n", + "An AutoML training pipeline is created with the `AutoMLTabularTrainingJob` class, with the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the `TrainingJob` resource.\n", + "- `optimization_prediction_type`: The type task to train the model for.\n", + " - `classification`: A tabuar classification model.\n", + " - `regression`: A tabular regression model.\n", + " - `forecasting`: A tabular forecasting model.\n", + "- `column_transformations`: (Optional): Transformations to apply to the input columns\n", + "- `optimization_objective`: The optimization objective to minimize or maximize.\n", + " - `minimize-log-loss`" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_automl_pipeline:tabular,lbn", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "dag = aip.AutoMLTabularTrainingJob(\n", + " display_name=\"bank_\" + TIMESTAMP,\n", + " optimization_prediction_type=\"classification\",\n", + " optimization_objective=\"minimize-log-loss\"\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "run_automl_pipeline:tabular", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "#### Run the training pipeline\n", + "\n", + "Next, you run the DAG to start the training job by invoking the method `run()`, with the following parameters:\n", + "\n", + "- `dataset`: The `Dataset` resource to train the model.\n", + "- `model_display_name`: The human readable name for the trained model.\n", + "- `target_column`: The name of the column to train as the label.\n", + "- `training_fraction_split`: The percentage of the dataset to use for training.\n", + "- `validation_fraction_split`: The percentage of the dataset to use for validation.\n", + "- `test_fraction_split`: The percentage of the dataset to use for test (holdout data).\n", + "- `budget_milli_node_hours`: (optional) Maximum training time specified in unit of millihours (1000 = hour).\n", + "- `disable_early_stopping`: If `True`, training maybe completed before using the entire budget if the service believes it cannot further improve on the model objective measurements.\n", + "\n", + "The `run` method when completed returns the `Model` resource.\n", + "\n", + "The execution of the training pipeline will take upto 20 minutes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "run_automl_pipeline:tabular", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "\n", + "model = dag.run(\n", + " dataset=dataset,\n", + " target_column=label_column,\n", + " model_display_name=\"bank_\" + TIMESTAMP,\n", + " training_fraction_split=0.6,\n", + " validation_fraction_split=0.2,\n", + " test_fraction_split=0.2,\n", + " budget_milli_node_hours=1000,\n", + " disable_early_stopping=False\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:mbsdk,dedicated", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "## Deploy the model\n", + "\n", + "Next, deploy your `Model` resource to an `Endpoint` resource for online prediction. To deploy the `Model` resource, you invoke the `deploy()` method. This call will create an `Endpoint` resource automatically.\n", + "\n", + "The method returns the created `Endpoint` resource.\n", + "\n", + "The `deploy()` method takes the following arguments:\n", + "- `machine_type`: The type of compute machine." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:mbsdk,dedicated", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "endpoint = model.deploy(machine_type=\"n1-standard-4\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction", + "repo": "snippets_common.ipynb" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_test_item:automl,online_prediction", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "### Make test item\n", + "\n", + "You will use synthetic data as a test data item. Don't be concerned that we are using synthetic data -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_test_item:automl,tabular,bank", + "repo": "snippets_automl.ipynb" + }, + "outputs": [], + "source": [ + "INSTANCE = {\"Age\": '58', \"Job\": \"managment\", \"MaritalStatus\": \"married\", \"Education\": \"teritary\", \"Default\": \"no\",\n", + " \"Balance\": '2143', \"Housing\": \"yes\", \"Loan\": \"no\", \"Contact\": \"unknown\", \"Day\": '5', \"Month\": \"may\",\n", + " \"Duration\": '261', \"Campaign\": '1', \"PDays\": '-1', \"Previous\": \"0\", \"POutcome\": \"unknown\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "predict_request:mbsdk,tabular,lbn", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Make the prediction\n", + "\n", + "Now that your `Model` resource is deployed to an `Endpoint` resource, one can do online predictions by sending prediction requests to the `Endpoint` resource.\n", + "\n", + "#### Request\n", + "\n", + "The format of each instance is:\n", + "\n", + " {[feature_list] }\n", + "\n", + "Since the `predict()` method can take multiple items (instances), send your single test item as a list of one test item.\n", + "\n", + "#### Response\n", + "\n", + "The response from the `predict()` call is a Python dictionary with the following entries:\n", + "\n", + "- `ids`: The internal assigned unique identifiers for each prediction request.\n", + "- TODO\n", + "- `deployed_model_id`: The Vertex identifier for the deployed `Model` resource which did the predictions." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "predict_request:mbsdk,tabular,lbn", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "instances_list = [INSTANCE]\n", + "\n", + "prediction = endpoint.predict(instances_list)\n", + "print(prediction)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "## Undeploy the model\n", + "\n", + "When you are done doing predictions, you undeploy the `Model` resource from the `Endpoint` resouce. This deprovisions all compute resources and ends billing for the deployed model." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "endpoint.undeploy_all()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "\n", + "# Delete the dataset using the Vertex dataset object\n", + "try:\n", + " if delete_dataset and 'dataset' in globals():\n", + " dataset.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex model object\n", + "try:\n", + " if delete_model and 'model' in globals():\n", + " model.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex endpoint object\n", + "try:\n", + " if delete_endpoint and 'model' in globals():\n", + " endpoint.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch prediction job using the Vertex batch prediction object\n", + "try:\n", + " if delete_batchjob and 'model' in globals():\n", + " batch_predict_job.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and 'BUCKET_NAME' in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.7.6" + }, + "vars": { + "AIP": "AI Platform", + "AUTOML": "AutoML", + "AUTOML_URL": "https://cloud.google.com/vertex-ai/docs/start/automl-users", + "BATCH_OR_ONLINE": "online", + "BQ": "BigQuery", + "COLAB": "Colab", + "DATASET_ALIAS": "bank", + "DATASET_NAME": "Bank Marketing", + "DATA_TYPE": "tabular", + "DOCKER": "Docker", + "EDGE": "Edge", + "EMAIL": "cdpe@gmail.com", + "GAPIC": "client library", + "GAPICP": "SDK", + "GCP": "Google Cloud", + "GCS": "Cloud Storage", + "GNOTEBOOK": "Google Cloud Notebook", + "HEIGHT": "128", + "IMPORT_FORMAT": "csv", + "MILLIHOURS": "1000", + "MODEL_TYPE": "tabular binary classification", + "NOTEBOOK": "sdk_automl_tabular_binary_classification_online.ipynb", + "REGION": "us-central1", + "REPO": "ai-platform-unified/notebooks/community/gapic/custom", + "TENSORFLOW": "TensorFlow", + "TFLite": "TFLite", + "TFServing": "TF Serving", + "TFV": "2-1", + "TITLE": "AutoML training tabular binary classification model for online prediction", + "TRAINING": "training", + "TRAINING_TIME": "20", + "TRAINING_WAIT": "60", + "WIDTH": "128", + "YEAR": "2021", + "null": "null", + "uCAIP": "Vertex" + } + }, + "nbformat": 4, + "nbformat_minor": 4 +} diff --git a/notebooks/community/sdk/sdk_automl_text_classification_batch.ipynb b/notebooks/community/sdk/sdk_automl_text_classification_batch.ipynb new file mode 100644 index 000000000..d34df2e7b --- /dev/null +++ b/notebooks/community/sdk/sdk_automl_text_classification_batch.ipynb @@ -0,0 +1,1170 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "# Vertex SDK: AutoML training text classification model for batch prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex SDK to create text classification models and do batch prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:happydb,tcn", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Happy Moments dataset](https://www.kaggle.com/ritresearch/happydb) from [Kaggle Datasets](https://www.kaggle.com/ritresearch/happydb). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,batch_prediction", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML text classification model from a Python script, and then do a batch prediction using the Vertex SDK. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Make a batch prediction.\n", + "\n", + "There is one key difference between using batch prediction and using online prediction:\n", + "\n", + "* Prediction Service: Does an on-demand prediction for the entire set of instances (i.e., one or more data items) and returns the results in real-time.\n", + "\n", + "* Batch Prediction Service: Does a queued (batch) prediction for the entire set of instances in the background and stores the results in a Cloud Storage bucket when ready." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex SDK." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "import sys\n", + "import os\n", + "\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = '--user'\n", + "else:\n", + " USER_FLAG = ''\n", + "\n", + "! pip3 install --upgrade google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" #@param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/ai-platform-unified/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "REGION = 'us-central1' #@param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you initialize the Vertex SDK for Python, you specify a Cloud Storage staging bucket. The staging bucket is where all the data associated with your dataset and model resources are retained across sessions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" #@param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "import google.cloud.aiplatform as aip" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "init_aip:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Initialize Vertex SDK\n", + "\n", + "Initialize the Vertex SDK for your project and corresponding bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "init_aip:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "aip.init(project=PROJECT_ID, staging_bucket=BUCKET_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML text classification model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_dataset", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "## Create a Dataset Resource\n", + "\n", + "First, you create an text Dataset resource for the Happy Moments dataset." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:text,u_dataset", + "repo": "snippets_common.ipynb" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for text has a couple of requirements for your text data.\n", + "\n", + "- Text examples must be stored in a CSV or JSONL file." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:tcn,u_dataset,csv", + "repo": "snippets_common.ipynb" + }, + "source": [ + "#### CSV\n", + "\n", + "For text classification, the CSV file has a few requirements:\n", + "\n", + "- No heading.\n", + "- First column is the text example or Cloud Storage path to text file (.txt suffix).\n", + "- Second column the label." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv", + "repo": "snippets_common.ipynb" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:happydb,csv,tcn", + "repo": "snippets_common.ipynb" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = 'gs://cloud-ml-data/NL-classification/happiness.csv'" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:csv", + "repo": "snippets_common.ipynb" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Happy Moments dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:csv", + "repo": "snippets_common.ipynb" + }, + "outputs": [], + "source": [ + "if 'IMPORT_FILES' in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_dataset:text,tcn", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Create the Dataset\n", + "\n", + "Next, create the `Dataset` resource using the `create()` method for the `TextDataset` class, which takes the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the `Dataset` resource.\n", + "- `gcs_source`: A list of one or more dataset index file to import the data items into the `Dataset` resource.\n", + "- `import_schema_uri`: The data labeling schema for the data items.\n", + "\n", + "This operation may take several minutes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_dataset:text,tcn", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "dataset = aip.TextDataset.create(\n", + " display_name=\"Happy Moments\" + \"_\" + TIMESTAMP,\n", + " gcs_source=[IMPORT_FILE],\n", + " import_schema_uri=aip.schema.dataset.ioformat.text.single_label_classification,\n", + ")\n", + "\n", + "print(dataset.resource_name)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML text classification model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_automl_pipeline:text,tcn", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Create and run training pipeline\n", + "\n", + "To train an AutoML text classification model, you perform two steps: 1) create a training pipeline, and 2) run the pipeline.\n", + "\n", + "#### Create training pipeline\n", + "\n", + "An AutoML training pipeline is created with the `AutoMLTextTrainingJob` class, with the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the `TrainingJob` resource.\n", + "- `prediction_type`: The type task to train the model for.\n", + " - `classification`: A text classification model.\n", + " - `sentiment`: A text sentiment analysis model.\n", + " - `extraction`: A text entity extraction model.\n", + "- `multi_label`: If a classification task, whether single (`False`) or multi-labeled (`True`).\n", + "\n", + "The instantiated object is the DAG for the training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_automl_pipeline:text,tcn", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "dag = aip.AutoMLTextTrainingJob(\n", + " display_name=\"happydb_\" + TIMESTAMP,\n", + " prediction_type=\"classification\",\n", + " multi_label=False\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "run_automl_pipeline:text", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "#### Run the training pipeline\n", + "\n", + "Next, you run the DAG to start the training job by invoking the method `run()`, with the following parameters:\n", + "\n", + "- `dataset`: The `Dataset` resource to train the model.\n", + "- `model_display_name`: The human readable name for the trained model.\n", + "- `training_fraction_split`: The percentage of the dataset to use for training.\n", + "- `validation_fraction_split`: The percentage of the dataset to use for validation.\n", + "- `test_fraction_split`: The percentage of the dataset to use for test (holdout data).\n", + "\n", + "The `run` method when completed returns the `Model` resource.\n", + "\n", + "The execution of the training pipeline will take upto 20 minutes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "run_automl_pipeline:text", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "model = dag.run(\n", + " dataset=dataset,\n", + " model_display_name=\"happydb_\" + TIMESTAMP,\n", + " training_fraction_split=0.8,\n", + " validation_fraction_split=0.1,\n", + " test_fraction_split=0.1\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy:batch_prediction", + "repo": "snippets_common.ipynb" + }, + "source": [ + "## Model deployment for batch prediction\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created for batch prediction. This differs from deploying a `Model` resource for online prediction.\n", + "\n", + "For online prediction, you:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource.\n", + "\n", + "3. Make online prediction requests to the `Endpoint` resource.\n", + "\n", + "For batch-prediction, you:\n", + "\n", + "1. Create a batch prediction job.\n", + "\n", + "2. The job service will provision resources for the batch prediction request.\n", + "\n", + "3. The results of the batch prediction request are returned to the caller.\n", + "\n", + "4. The job service will unprovision the resoures for the batch prediction request." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction", + "repo": "snippets_common.ipynb" + }, + "source": [ + "## Make a batch prediction request\n", + "\n", + "Now do a batch prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_items:batch_prediction", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "### Get test item(s)\n", + "\n", + "Now do a batch prediction to your Vertex model. You will use arbitrary examples out of the dataset as a test items. Don't be concerned that the examples were likely used in training the model -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_items:automl,tcn,csv", + "repo": "snippets_automl.ipynb" + }, + "outputs": [], + "source": [ + "test_items = ! gsutil cat $IMPORT_FILE | head -n2\n", + "if len(test_items[0]) == 3:\n", + " _, test_item_1, test_label_1 = str(test_items[0]).split(',')\n", + " _, test_item_2, test_label_2 = str(test_items[1]).split(',')\n", + "else:\n", + " test_item_1, test_label_1 = str(test_items[0]).split(',')\n", + " test_item_2, test_label_2 = str(test_items[1]).split(',')\n", + "\n", + "print(test_item_1, test_label_1)\n", + "print(test_item_2, test_label_2)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "copy_test_items:batch_prediction", + "repo": "snippets_common.ipynb" + }, + "source": [ + "### Copy test item(s)\n", + "\n", + "For the batch prediction, you will copy the test items over to your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copy_test_items:batch_prediction", + "repo": "snippets_common.ipynb" + }, + "outputs": [], + "source": [ + "file_1 = test_item_1.split('/')[-1]\n", + "file_2 = test_item_2.split('/')[-1]\n", + "\n", + "! gsutil cp $test_item_1 $BUCKET_NAME/$file_1\n", + "! gsutil cp $test_item_2 $BUCKET_NAME/$file_2\n", + "\n", + "test_item_1 = BUCKET_NAME + \"/\" + file_1\n", + "test_item_2 = BUCKET_NAME + \"/\" + file_2" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:automl,text", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "Now make a batch input file, which you will store in your local Cloud Storage bucket. The batch input file can only be in JSONL format. For JSONL file, you make one dictionary entry per line for each data item (instance). The dictionary contains the key/value pairs:\n", + "\n", + "- `content`: The Cloud Storage path to the file with the text item.\n", + "- `mime_type`: The content type. In our example, it is an `text` file.\n", + "\n", + "For example:\n", + "\n", + " {'content': '[your-bucket]/file1.txt', 'mime_type': 'text'}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_file:automl,text", + "repo": "snippets_automl.ipynb" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "import json\n", + "\n", + "gcs_test_item_1 = BUCKET_NAME + '/test1.txt'\n", + "with tf.io.gfile.GFile(gcs_test_item_1, 'w') as f:\n", + " f.write(test_item_1 + '\\n')\n", + "gcs_test_item_2 = BUCKET_NAME + '/test2.txt'\n", + "with tf.io.gfile.GFile(gcs_test_item_2, 'w') as f:\n", + " f.write(test_item_2 + '\\n')\n", + "\n", + "gcs_input_uri = BUCKET_NAME + '/test.jsonl'\n", + "with tf.io.gfile.GFile(gcs_input_uri, 'w') as f:\n", + " data = {\"content\": gcs_test_item_1, \"mime_type\": \"text/plain\"}\n", + " f.write(json.dumps(data) + '\\n')\n", + " data = {\"content\": gcs_test_item_2, \"mime_type\": \"text/plain\"}\n", + " f.write(json.dumps(data) + '\\n')\n", + "\n", + "print(gcs_input_uri)\n", + "! gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_request:mbsdk,automl", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Make the batch prediction request\n", + "\n", + "Now that your `Model` resource is trained, you can make a batch prediction by invoking the `batch_request()` method, with the following parameters:\n", + "\n", + "- `job_display_name`: The human readable name for the batch prediction job.\n", + "- `gcs_source`: A list of one or more batch request input files.\n", + "- `gcs_destination_prefix`: The Cloud Storage location for storing the batch prediction resuls.\n", + "- `sync`: If set to `True`, the call will block while waiting for the asynchronous batch job to complete." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_request:mbsdk,automl", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "batch_predict_job = model.batch_predict(\n", + " job_display_name=\"$(DATASET_ALIAS)_\" + TIMESTAMP,\n", + " gcs_source=gcs_input_uri,\n", + " gcs_destination_prefix=BUCKET_NAME,\n", + " sync=False\n", + ")\n", + "\n", + "print(batch_predict_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_batch_job:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Wait for completion of batch prediction job\n", + "\n", + "Next, wait for the batch job to complete." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_batch_job:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "batch_predict_job.wait()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_batch_prediction:mbsdk,tcn", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Get the predictions\n", + "\n", + "Next, get the results from the completed batch prediction job.\n", + "\n", + "The results are written to the Cloud Storage output bucket you specified in the batch prediction request. You call the method `iter_outputs()` to get a list of each Cloud Storage file generated with the results. Each file contains one or more prediction requests in a JSON format:\n", + "\n", + "- `content`: The prediction request.\n", + "- `prediction`: The prediction response.\n", + " - `ids`: The internal assigned unique identifiers for each prediction request.\n", + " - `displayNames`: The class names for each class label.\n", + " - `confidences`: The predicted confidence, between 0 and 1, per class label." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_batch_prediction:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "bp_iter_outputs = batch_predict_job.iter_outputs()\n", + "\n", + "prediction_results = list()\n", + "for blob in bp_iter_outputs:\n", + " if blob.name.split(\"/\")[-1].startswith(\"prediction\"):\n", + " prediction_results.append(blob.name)\n", + "\n", + "tags = list()\n", + "for prediction_result in prediction_results:\n", + " gfile_name = f\"gs://{bp_iter_outputs.bucket.name}/{prediction_result}\"\n", + " with tf.io.gfile.GFile(name=gfile_name, mode=\"r\") as gfile:\n", + " for line in gfile.readlines():\n", + " line = json.loads(line)\n", + " print(line)\n", + " break" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "\n", + "# Delete the dataset using the Vertex dataset object\n", + "try:\n", + " if delete_dataset and 'dataset' in globals():\n", + " dataset.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex model object\n", + "try:\n", + " if delete_model and 'model' in globals():\n", + " model.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex endpoint object\n", + "try:\n", + " if delete_endpoint and 'model' in globals():\n", + " endpoint.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch prediction job using the Vertex batch prediction object\n", + "try:\n", + " if delete_batchjob and 'model' in globals():\n", + " batch_predict_job.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and 'BUCKET_NAME' in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.7.6" + }, + "vars": { + "AIP": "AI Platform", + "AUTOML": "AutoML", + "AUTOML_URL": "https://cloud.google.com/vertex-ai/docs/start/automl-users", + "BATCH_OR_ONLINE": "batch", + "BQ": "BigQuery", + "COLAB": "Colab", + "DATASET_ALIAS": "happydb", + "DATASET_NAME": "Happy Moments", + "DATA_TYPE": "text", + "DOCKER": "Docker", + "EDGE": "Edge", + "EMAIL": "cdpe@gmail.com", + "GAPIC": "client library", + "GAPICP": "SDK", + "GCP": "Google Cloud", + "GCS": "Cloud Storage", + "GNOTEBOOK": "Google Cloud Notebook", + "HEIGHT": "128", + "IMPORT_FORMAT": "csv", + "MODEL_TYPE": "text classification", + "NOTEBOOK": "sdk_automl_text_classification_batch.ipynb", + "REGION": "us-central1", + "REPO": "ai-platform-unified/notebooks/community/gapic/custom", + "TENSORFLOW": "TensorFlow", + "TFLite": "TFLite", + "TFServing": "TF Serving", + "TFV": "2-1", + "TITLE": "AutoML training text classification model for batch prediction", + "TRAINING": "training", + "TRAINING_TIME": "20", + "TRAINING_WAIT": "60", + "WIDTH": "128", + "YEAR": "2021", + "null": "null", + "uCAIP": "Vertex" + } + }, + "nbformat": 4, + "nbformat_minor": 4 +} diff --git a/notebooks/community/sdk/sdk_automl_text_classification_online.ipynb b/notebooks/community/sdk/sdk_automl_text_classification_online.ipynb new file mode 100644 index 000000000..3db78f113 --- /dev/null +++ b/notebooks/community/sdk/sdk_automl_text_classification_online.ipynb @@ -0,0 +1,1046 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "# Vertex SDK: AutoML training text classification model for online prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex SDK to create text classification models and do online prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:happydb,tcn", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Happy Moments dataset](https://www.kaggle.com/ritresearch/happydb) from [Kaggle Datasets](https://www.kaggle.com/ritresearch/happydb). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,online_prediction", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML text classification model and deploy for online prediction from a Python script using the Vertex SDK. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model`." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex SDK." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "import sys\n", + "import os\n", + "\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = '--user'\n", + "else:\n", + " USER_FLAG = ''\n", + "\n", + "! pip3 install --upgrade google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" #@param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/ai-platform-unified/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "REGION = 'us-central1' #@param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you initialize the Vertex SDK for Python, you specify a Cloud Storage staging bucket. The staging bucket is where all the data associated with your dataset and model resources are retained across sessions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" #@param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "import google.cloud.aiplatform as aip" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "init_aip:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Initialize Vertex SDK\n", + "\n", + "Initialize the Vertex SDK for your project and corresponding bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "init_aip:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "aip.init(project=PROJECT_ID, staging_bucket=BUCKET_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML text classification model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_dataset", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "## Create a Dataset Resource\n", + "\n", + "First, you create an text Dataset resource for the Happy Moments dataset." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:text,u_dataset", + "repo": "snippets_common.ipynb" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for text has a couple of requirements for your text data.\n", + "\n", + "- Text examples must be stored in a CSV or JSONL file." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:tcn,u_dataset,csv", + "repo": "snippets_common.ipynb" + }, + "source": [ + "#### CSV\n", + "\n", + "For text classification, the CSV file has a few requirements:\n", + "\n", + "- No heading.\n", + "- First column is the text example or Cloud Storage path to text file (.txt suffix).\n", + "- Second column the label." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv", + "repo": "snippets_common.ipynb" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:happydb,csv,tcn", + "repo": "snippets_common.ipynb" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = 'gs://cloud-ml-data/NL-classification/happiness.csv'" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:csv", + "repo": "snippets_common.ipynb" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Happy Moments dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:csv", + "repo": "snippets_common.ipynb" + }, + "outputs": [], + "source": [ + "if 'IMPORT_FILES' in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_dataset:text,tcn", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Create the Dataset\n", + "\n", + "Next, create the `Dataset` resource using the `create()` method for the `TextDataset` class, which takes the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the `Dataset` resource.\n", + "- `gcs_source`: A list of one or more dataset index file to import the data items into the `Dataset` resource.\n", + "- `import_schema_uri`: The data labeling schema for the data items.\n", + "\n", + "This operation may take several minutes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_dataset:text,tcn", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "dataset = aip.TextDataset.create(\n", + " display_name=\"Happy Moments\" + \"_\" + TIMESTAMP,\n", + " gcs_source=[IMPORT_FILE],\n", + " import_schema_uri=aip.schema.dataset.ioformat.text.single_label_classification,\n", + ")\n", + "\n", + "print(dataset.resource_name)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML text classification model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_automl_pipeline:text,tcn", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Create and run training pipeline\n", + "\n", + "To train an AutoML text classification model, you perform two steps: 1) create a training pipeline, and 2) run the pipeline.\n", + "\n", + "#### Create training pipeline\n", + "\n", + "An AutoML training pipeline is created with the `AutoMLTextTrainingJob` class, with the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the `TrainingJob` resource.\n", + "- `prediction_type`: The type task to train the model for.\n", + " - `classification`: A text classification model.\n", + " - `sentiment`: A text sentiment analysis model.\n", + " - `extraction`: A text entity extraction model.\n", + "- `multi_label`: If a classification task, whether single (`False`) or multi-labeled (`True`).\n", + "\n", + "The instantiated object is the DAG for the training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_automl_pipeline:text,tcn", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "dag = aip.AutoMLTextTrainingJob(\n", + " display_name=\"happydb_\" + TIMESTAMP,\n", + " prediction_type=\"classification\",\n", + " multi_label=False\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "run_automl_pipeline:text", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "#### Run the training pipeline\n", + "\n", + "Next, you run the DAG to start the training job by invoking the method `run()`, with the following parameters:\n", + "\n", + "- `dataset`: The `Dataset` resource to train the model.\n", + "- `model_display_name`: The human readable name for the trained model.\n", + "- `training_fraction_split`: The percentage of the dataset to use for training.\n", + "- `validation_fraction_split`: The percentage of the dataset to use for validation.\n", + "- `test_fraction_split`: The percentage of the dataset to use for test (holdout data).\n", + "\n", + "The `run` method when completed returns the `Model` resource.\n", + "\n", + "The execution of the training pipeline will take upto 20 minutes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "run_automl_pipeline:text", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "model = dag.run(\n", + " dataset=dataset,\n", + " model_display_name=\"happydb_\" + TIMESTAMP,\n", + " training_fraction_split=0.8,\n", + " validation_fraction_split=0.1,\n", + " test_fraction_split=0.1\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "## Deploy the model\n", + "\n", + "Next, deploy your `Model` resource to an `Endpoint` resource for online prediction. To deploy the `Model` resource, you invoke the `deploy()` method. This call will create an `Endpoint` resource automatically.\n", + "\n", + "The method returns the created `Endpoint` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "endpoint = model.deploy()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction", + "repo": "snippets_common.ipynb" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_item", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "### Get test item\n", + "\n", + "You will use an arbitrary example out of the dataset as a test item. Don't be concerned that the example was likely used in training the model -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:automl,tcn,csv", + "repo": "snippets_automl.ipynb" + }, + "outputs": [], + "source": [ + "test_item = ! gsutil cat $IMPORT_FILE | head -n1\n", + "if len(test_item[0]) == 3:\n", + " _, test_item, test_label = str(test_item[0]).split(',')\n", + "else:\n", + " test_item, test_label = str(test_item[0]).split(',')\n", + "\n", + "print(test_item, test_label)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "predict_request:mbsdk,text,tcn", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Make the prediction\n", + "\n", + "Now that your `Model` resource is deployed to an `Endpoint` resource, one can do online predictions by sending prediction requests to the `Endpoint` resource.\n", + "\n", + "#### Request\n", + "\n", + "The format of each instance is:\n", + "\n", + " { 'content': { [text_string] } }\n", + "\n", + "Since the `predict()` method can take multiple items (instances), send your single test item as a list of one test item.\n", + "\n", + "#### Response\n", + "\n", + "The response from the `predict()` call is a Python dictionary with the following entries:\n", + "\n", + "- `ids`: The internal assigned unique identifiers for each prediction request.\n", + "- `displayNames`: The class names for each class label.\n", + "- `confidences`: The predicted confidence, between 0 and 1, per class label.\n", + "- `deployed_model_id`: The Vertex identifier for the deployed `Model` resource which did the predictions." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "predict_request:mbsdk,text,tcn", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "instances_list = [{\"content\": test_item}]\n", + "\n", + "prediction = endpoint.predict(instances_list)\n", + "print(prediction)\n", + "\n", + "prediction_instance = prediction.predictions[0]\n", + "\n", + "confidences = prediction_instance[\"confidences\"]\n", + "max_index = confidences.index(max(confidences))\n", + "print(prediction_instance[\"displayNames\"][max_index])" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "## Undeploy the model\n", + "\n", + "When you are done doing predictions, you undeploy the `Model` resource from the `Endpoint` resouce. This deprovisions all compute resources and ends billing for the deployed model." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "endpoint.undeploy_all()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "\n", + "# Delete the dataset using the Vertex dataset object\n", + "try:\n", + " if delete_dataset and 'dataset' in globals():\n", + " dataset.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex model object\n", + "try:\n", + " if delete_model and 'model' in globals():\n", + " model.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex endpoint object\n", + "try:\n", + " if delete_endpoint and 'model' in globals():\n", + " endpoint.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch prediction job using the Vertex batch prediction object\n", + "try:\n", + " if delete_batchjob and 'model' in globals():\n", + " batch_predict_job.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and 'BUCKET_NAME' in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.7.6" + }, + "vars": { + "AIP": "AI Platform", + "AUTOML": "AutoML", + "AUTOML_URL": "https://cloud.google.com/vertex-ai/docs/start/automl-users", + "BATCH_OR_ONLINE": "online", + "BQ": "BigQuery", + "COLAB": "Colab", + "DATASET_ALIAS": "happydb", + "DATASET_NAME": "Happy Moments", + "DATA_TYPE": "text", + "DOCKER": "Docker", + "EDGE": "Edge", + "EMAIL": "cdpe@gmail.com", + "GAPIC": "client library", + "GAPICP": "SDK", + "GCP": "Google Cloud", + "GCS": "Cloud Storage", + "GNOTEBOOK": "Google Cloud Notebook", + "IMPORT_FORMAT": "csv", + "MILLIHOURS": "20000", + "MODEL_TYPE": "text classification", + "NOTEBOOK": "sdk_automl_text_classification_online.ipynb", + "REGION": "us-central1", + "REPO": "ai-platform-unified/notebooks/community/gapic/custom", + "TENSORFLOW": "TensorFlow", + "TFLite": "TFLite", + "TFServing": "TF Serving", + "TFV": "2-1", + "TITLE": "AutoML training text classification model for online prediction", + "TRAINING": "training", + "TRAINING_TIME": "20", + "TRAINING_WAIT": "60", + "YEAR": "2021", + "null": "null", + "uCAIP": "Vertex" + } + }, + "nbformat": 4, + "nbformat_minor": 4 +} diff --git a/notebooks/community/sdk/sdk_automl_text_sentiment_analysis_batch.ipynb b/notebooks/community/sdk/sdk_automl_text_sentiment_analysis_batch.ipynb new file mode 100644 index 000000000..a22ecbe44 --- /dev/null +++ b/notebooks/community/sdk/sdk_automl_text_sentiment_analysis_batch.ipynb @@ -0,0 +1,1172 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "# Vertex SDK: AutoML training text sentiment analysis model for batch prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex SDK to create text sentiment analysis models and do batch prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:claritin,tst", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Crowdflower Claritin-Twitter dataset](https://data.world/crowdflower/claritin-twitter) from [data.world Datasets](https://data.world). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,batch_prediction", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML text sentiment analysis model from a Python script, and then do a batch prediction using the Vertex SDK. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Make a batch prediction.\n", + "\n", + "There is one key difference between using batch prediction and using online prediction:\n", + "\n", + "* Prediction Service: Does an on-demand prediction for the entire set of instances (i.e., one or more data items) and returns the results in real-time.\n", + "\n", + "* Batch Prediction Service: Does a queued (batch) prediction for the entire set of instances in the background and stores the results in a Cloud Storage bucket when ready." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex SDK." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "import sys\n", + "import os\n", + "\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = '--user'\n", + "else:\n", + " USER_FLAG = ''\n", + "\n", + "! pip3 install --upgrade google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" #@param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/ai-platform-unified/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "REGION = 'us-central1' #@param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you initialize the Vertex SDK for Python, you specify a Cloud Storage staging bucket. The staging bucket is where all the data associated with your dataset and model resources are retained across sessions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" #@param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "import google.cloud.aiplatform as aip" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "init_aip:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Initialize Vertex SDK\n", + "\n", + "Initialize the Vertex SDK for your project and corresponding bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "init_aip:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "aip.init(project=PROJECT_ID, staging_bucket=BUCKET_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML text sentiment analysis model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_dataset", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "## Create a Dataset Resource\n", + "\n", + "First, you create an text Dataset resource for the Crowdflower Claritin-Twitter dataset." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:text,u_dataset", + "repo": "snippets_common.ipynb" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for text has a couple of requirements for your text data.\n", + "\n", + "- Text examples must be stored in a CSV or JSONL file." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:tst,u_dataset,csv", + "repo": "snippets_common.ipynb" + }, + "source": [ + "#### CSV\n", + "\n", + "For text sentiment analysis, the CSV file has a few requirements:\n", + "\n", + "- No heading.\n", + "- First column is the text example or Cloud Storage path to text file.\n", + "- Second column the label (i.e., sentiment).\n", + "- Third column is the maximum sentiment value. For example, if the range is 0 to 3, then the maximum value is 3." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv", + "repo": "snippets_common.ipynb" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:claritin,csv,tst", + "repo": "snippets_common.ipynb" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = 'gs://cloud-samples-data/language/claritin.csv'\n", + "SENTIMENT_MAX = 4" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:csv", + "repo": "snippets_common.ipynb" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Crowdflower Claritin-Twitter dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:csv", + "repo": "snippets_common.ipynb" + }, + "outputs": [], + "source": [ + "if 'IMPORT_FILES' in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_dataset:text,tst", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Create the Dataset\n", + "\n", + "Next, create the `Dataset` resource using the `create()` method for the `TextDataset` class, which takes the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the `Dataset` resource.\n", + "- `gcs_source`: A list of one or more dataset index file to import the data items into the `Dataset` resource.\n", + "- `import_schema_uri`: The data labeling schema for the data items.\n", + "\n", + "This operation may take several minutes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_dataset:text,tst", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "dataset = aip.TextDataset.create(\n", + " display_name=\"Crowdflower Claritin-Twitter\" + \"_\" + TIMESTAMP,\n", + " gcs_source=[IMPORT_FILE],\n", + " import_schema_uri=aip.schema.dataset.ioformat.text.sentiment,\n", + ")\n", + "\n", + "print(dataset.resource_name)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML text sentiment analysis model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_automl_pipeline:text,tst", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Create and run training pipeline\n", + "\n", + "To train an AutoML text sentiment analysis model, you perform two steps: 1) create a training pipeline, and 2) run the pipeline.\n", + "\n", + "#### Create training pipeline\n", + "\n", + "An AutoML training pipeline is created with the `AutoMLTextTrainingJob` class, with the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the `TrainingJob` resource.\n", + "- `prediction_type`: The type task to train the model for.\n", + " - `classification`: A text classification model.\n", + " - `sentiment`: A text sentiment analysis model.\n", + " - `extraction`: A text entity extraction model.\n", + "- `sentiment_max`: The maximum sentiment value.\n", + "\n", + "The instantiated object is the DAG for the training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_automl_pipeline:text,tst", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "dag = aip.AutoMLTextTrainingJob(\n", + " display_name=\"claritin_\" + TIMESTAMP,\n", + " prediction_type=\"sentiment\",\n", + " sentiment_max=SENTIMENT_MAX\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "run_automl_pipeline:text", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "#### Run the training pipeline\n", + "\n", + "Next, you run the DAG to start the training job by invoking the method `run()`, with the following parameters:\n", + "\n", + "- `dataset`: The `Dataset` resource to train the model.\n", + "- `model_display_name`: The human readable name for the trained model.\n", + "- `training_fraction_split`: The percentage of the dataset to use for training.\n", + "- `validation_fraction_split`: The percentage of the dataset to use for validation.\n", + "- `test_fraction_split`: The percentage of the dataset to use for test (holdout data).\n", + "\n", + "The `run` method when completed returns the `Model` resource.\n", + "\n", + "The execution of the training pipeline will take upto 20 minutes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "run_automl_pipeline:text", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "model = dag.run(\n", + " dataset=dataset,\n", + " model_display_name=\"claritin_\" + TIMESTAMP,\n", + " training_fraction_split=0.8,\n", + " validation_fraction_split=0.1,\n", + " test_fraction_split=0.1\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy:batch_prediction", + "repo": "snippets_common.ipynb" + }, + "source": [ + "## Model deployment for batch prediction\n", + "\n", + "Now deploy the trained Vertex `Model` resource you created for batch prediction. This differs from deploying a `Model` resource for online prediction.\n", + "\n", + "For online prediction, you:\n", + "\n", + "1. Create an `Endpoint` resource for deploying the `Model` resource to.\n", + "\n", + "2. Deploy the `Model` resource to the `Endpoint` resource.\n", + "\n", + "3. Make online prediction requests to the `Endpoint` resource.\n", + "\n", + "For batch-prediction, you:\n", + "\n", + "1. Create a batch prediction job.\n", + "\n", + "2. The job service will provision resources for the batch prediction request.\n", + "\n", + "3. The results of the batch prediction request are returned to the caller.\n", + "\n", + "4. The job service will unprovision the resoures for the batch prediction request." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction", + "repo": "snippets_common.ipynb" + }, + "source": [ + "## Make a batch prediction request\n", + "\n", + "Now do a batch prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_items:batch_prediction", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "### Get test item(s)\n", + "\n", + "Now do a batch prediction to your Vertex model. You will use arbitrary examples out of the dataset as a test items. Don't be concerned that the examples were likely used in training the model -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_items:automl,tst,csv", + "repo": "snippets_automl.ipynb" + }, + "outputs": [], + "source": [ + "test_items = ! gsutil cat $IMPORT_FILE | head -n2\n", + "\n", + "if len(test_items[0]) == 4:\n", + " _, test_item_1, test_label_1, _ = str(test_items[0]).split(',')\n", + " _, test_item_2, test_label_2, _ = str(test_items[1]).split(',')\n", + "else:\n", + " test_item_1, test_label_1, _ = str(test_items[0]).split(',')\n", + " test_item_2, test_label_2, _ = str(test_items[1]).split(',')\n", + "\n", + "\n", + "print(test_item_1, test_label_1)\n", + "print(test_item_2, test_label_2)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "copy_test_items:batch_prediction", + "repo": "snippets_common.ipynb" + }, + "source": [ + "### Copy test item(s)\n", + "\n", + "For the batch prediction, you will copy the test items over to your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copy_test_items:batch_prediction", + "repo": "snippets_common.ipynb" + }, + "outputs": [], + "source": [ + "file_1 = test_item_1.split('/')[-1]\n", + "file_2 = test_item_2.split('/')[-1]\n", + "\n", + "! gsutil cp $test_item_1 $BUCKET_NAME/$file_1\n", + "! gsutil cp $test_item_2 $BUCKET_NAME/$file_2\n", + "\n", + "test_item_1 = BUCKET_NAME + \"/\" + file_1\n", + "test_item_2 = BUCKET_NAME + \"/\" + file_2" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_file:automl,text", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "### Make the batch input file\n", + "\n", + "Now make a batch input file, which you will store in your local Cloud Storage bucket. The batch input file can only be in JSONL format. For JSONL file, you make one dictionary entry per line for each data item (instance). The dictionary contains the key/value pairs:\n", + "\n", + "- `content`: The Cloud Storage path to the file with the text item.\n", + "- `mime_type`: The content type. In our example, it is an `text` file.\n", + "\n", + "For example:\n", + "\n", + " {'content': '[your-bucket]/file1.txt', 'mime_type': 'text'}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_file:automl,text", + "repo": "snippets_automl.ipynb" + }, + "outputs": [], + "source": [ + "import tensorflow as tf\n", + "import json\n", + "\n", + "gcs_test_item_1 = BUCKET_NAME + '/test1.txt'\n", + "with tf.io.gfile.GFile(gcs_test_item_1, 'w') as f:\n", + " f.write(test_item_1 + '\\n')\n", + "gcs_test_item_2 = BUCKET_NAME + '/test2.txt'\n", + "with tf.io.gfile.GFile(gcs_test_item_2, 'w') as f:\n", + " f.write(test_item_2 + '\\n')\n", + "\n", + "gcs_input_uri = BUCKET_NAME + '/test.jsonl'\n", + "with tf.io.gfile.GFile(gcs_input_uri, 'w') as f:\n", + " data = {\"content\": gcs_test_item_1, \"mime_type\": \"text/plain\"}\n", + " f.write(json.dumps(data) + '\\n')\n", + " data = {\"content\": gcs_test_item_2, \"mime_type\": \"text/plain\"}\n", + " f.write(json.dumps(data) + '\\n')\n", + "\n", + "print(gcs_input_uri)\n", + "! gsutil cat $gcs_input_uri" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_batch_request:mbsdk,automl", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Make the batch prediction request\n", + "\n", + "Now that your `Model` resource is trained, you can make a batch prediction by invoking the `batch_request()` method, with the following parameters:\n", + "\n", + "- `job_display_name`: The human readable name for the batch prediction job.\n", + "- `gcs_source`: A list of one or more batch request input files.\n", + "- `gcs_destination_prefix`: The Cloud Storage location for storing the batch prediction resuls.\n", + "- `sync`: If set to `True`, the call will block while waiting for the asynchronous batch job to complete." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "make_batch_request:mbsdk,automl", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "batch_predict_job = model.batch_predict(\n", + " job_display_name=\"$(DATASET_ALIAS)_\" + TIMESTAMP,\n", + " gcs_source=gcs_input_uri,\n", + " gcs_destination_prefix=BUCKET_NAME,\n", + " sync=False\n", + ")\n", + "\n", + "print(batch_predict_job)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "wait_batch_job:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Wait for completion of batch prediction job\n", + "\n", + "Next, wait for the batch job to complete." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "wait_batch_job:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "batch_predict_job.wait()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_batch_prediction:mbsdk,tst", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Get the predictions\n", + "\n", + "Next, get the results from the completed batch prediction job.\n", + "\n", + "The results are written to the Cloud Storage output bucket you specified in the batch prediction request. You call the method `iter_outputs()` to get a list of each Cloud Storage file generated with the results. Each file contains one or more prediction requests in a JSON format:\n", + "\n", + "- `content`: The prediction request.\n", + "- `prediction`: The prediction response.\n", + " - `sentiment`: The sentiment value." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_batch_prediction:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "bp_iter_outputs = batch_predict_job.iter_outputs()\n", + "\n", + "prediction_results = list()\n", + "for blob in bp_iter_outputs:\n", + " if blob.name.split(\"/\")[-1].startswith(\"prediction\"):\n", + " prediction_results.append(blob.name)\n", + "\n", + "tags = list()\n", + "for prediction_result in prediction_results:\n", + " gfile_name = f\"gs://{bp_iter_outputs.bucket.name}/{prediction_result}\"\n", + " with tf.io.gfile.GFile(name=gfile_name, mode=\"r\") as gfile:\n", + " for line in gfile.readlines():\n", + " line = json.loads(line)\n", + " print(line)\n", + " break" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "\n", + "# Delete the dataset using the Vertex dataset object\n", + "try:\n", + " if delete_dataset and 'dataset' in globals():\n", + " dataset.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex model object\n", + "try:\n", + " if delete_model and 'model' in globals():\n", + " model.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex endpoint object\n", + "try:\n", + " if delete_endpoint and 'model' in globals():\n", + " endpoint.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch prediction job using the Vertex batch prediction object\n", + "try:\n", + " if delete_batchjob and 'model' in globals():\n", + " batch_predict_job.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and 'BUCKET_NAME' in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.7.6" + }, + "vars": { + "AIP": "AI Platform", + "AUTOML": "AutoML", + "AUTOML_URL": "https://cloud.google.com/vertex-ai/docs/start/automl-users", + "BATCH_OR_ONLINE": "batch", + "BQ": "BigQuery", + "COLAB": "Colab", + "DATASET_ALIAS": "claritin", + "DATASET_NAME": "Crowdflower Claritin-Twitter", + "DATA_TYPE": "text", + "DOCKER": "Docker", + "EDGE": "Edge", + "EMAIL": "cdpe@gmail.com", + "GAPIC": "client library", + "GAPICP": "SDK", + "GCP": "Google Cloud", + "GCS": "Cloud Storage", + "GNOTEBOOK": "Google Cloud Notebook", + "HEIGHT": "128", + "IMPORT_FORMAT": "csv", + "MODEL_TYPE": "text sentiment analysis", + "NOTEBOOK": "sdk_automl_text_sentiment_analysis_batch.ipynb", + "REGION": "us-central1", + "REPO": "ai-platform-unified/notebooks/community/gapic/custom", + "TENSORFLOW": "TensorFlow", + "TFLite": "TFLite", + "TFServing": "TF Serving", + "TFV": "2-1", + "TITLE": "AutoML training text sentiment analysis model for batch prediction", + "TRAINING": "training", + "TRAINING_TIME": "20", + "TRAINING_WAIT": "60", + "WIDTH": "128", + "YEAR": "2021", + "null": "null", + "uCAIP": "Vertex" + } + }, + "nbformat": 4, + "nbformat_minor": 4 +} diff --git a/notebooks/community/sdk/sdk_automl_text_sentiment_analysis_online.ipynb b/notebooks/community/sdk/sdk_automl_text_sentiment_analysis_online.ipynb new file mode 100644 index 000000000..44f0e33b1 --- /dev/null +++ b/notebooks/community/sdk/sdk_automl_text_sentiment_analysis_online.ipynb @@ -0,0 +1,1041 @@ +{ + "cells": [ + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "copyright", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "# Copyright 2021 Google LLC\n", + "#\n", + "# Licensed under the Apache License, Version 2.0 (the \"License\");\n", + "# you may not use this file except in compliance with the License.\n", + "# You may obtain a copy of the License at\n", + "#\n", + "# https://www.apache.org/licenses/LICENSE-2.0\n", + "#\n", + "# Unless required by applicable law or agreed to in writing, software\n", + "# distributed under the License is distributed on an \"AS IS\" BASIS,\n", + "# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.\n", + "# See the License for the specific language governing permissions and\n", + "# limitations under the License." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "title", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "# Vertex SDK: AutoML training text sentiment analysis model for online prediction\n", + "\n", + "\n", + " \n", + " \n", + "
\n", + " \n", + " \"Colab Run in Colab\n", + " \n", + " \n", + " \n", + " \"GitHub\n", + " View on GitHub\n", + " \n", + "
\n", + "


" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "overview:automl", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Overview\n", + "\n", + "\n", + "This tutorial demonstrates how to use the Vertex SDK to create text sentiment analysis models and do online prediction using Google Cloud's [AutoML](https://cloud.google.com/vertex-ai/docs/start/automl-users)." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "dataset:claritin,tst", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Dataset\n", + "\n", + "The dataset used for this tutorial is the [Crowdflower Claritin-Twitter dataset](https://data.world/crowdflower/claritin-twitter) from [data.world Datasets](https://data.world). The version of the dataset you will use in this tutorial is stored in a public Cloud Storage bucket." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "objective:automl,training,online_prediction", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Objective\n", + "\n", + "In this tutorial, you create an AutoML text sentiment analysis model and deploy for online prediction from a Python script using the Vertex SDK. You can alternatively create and deploy models using the `gcloud` command-line tool or online using the Google Cloud Console.\n", + "\n", + "The steps performed include:\n", + "\n", + "- Create a Vertex `Dataset` resource.\n", + "- Train the model.\n", + "- View the model evaluation.\n", + "- Deploy the `Model` resource to a serving `Endpoint` resource.\n", + "- Make a prediction.\n", + "- Undeploy the `Model`." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "costs", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Costs\n", + "\n", + "This tutorial uses billable components of Google Cloud (GCP):\n", + "\n", + "* Vertex AI\n", + "* Cloud Storage\n", + "\n", + "Learn about [Vertex AI\n", + "pricing](https://cloud.google.com/vertex-ai/pricing) and [Cloud Storage\n", + "pricing](https://cloud.google.com/storage/pricing), and use the [Pricing\n", + "Calculator](https://cloud.google.com/products/calculator/)\n", + "to generate a cost estimate based on your projected usage." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_aip", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Installation\n", + "\n", + "Install the latest version of Vertex SDK." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_aip", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "import sys\n", + "import os\n", + "\n", + "\n", + "# Google Cloud Notebook\n", + "if os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " USER_FLAG = '--user'\n", + "else:\n", + " USER_FLAG = ''\n", + "\n", + "! pip3 install --upgrade google-cloud-aiplatform $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "install_storage", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "Install the latest GA version of *google-cloud-storage* library as well." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "install_storage", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! pip3 install -U google-cloud-storage $USER_FLAG" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "restart", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Restart the kernel\n", + "\n", + "Once you've installed the Vertex SDK and Google *cloud-storage*, you need to restart the notebook kernel so it can find the packages." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "restart", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "if not os.getenv(\"IS_TESTING\"):\n", + " # Automatically restart kernel after installs\n", + " import IPython\n", + " app = IPython.Application.instance()\n", + " app.kernel.do_shutdown(True)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "before_you_begin", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Before you begin\n", + "\n", + "### GPU runtime\n", + "\n", + "*Make sure you're running this notebook in a GPU runtime if you have that option. In Colab, select* **Runtime > Change Runtime Type > GPU**\n", + "\n", + "### Set up your Google Cloud project\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "1. [Select or create a Google Cloud project](https://console.cloud.google.com/cloud-resource-manager). When you first create an account, you get a $300 free credit towards your compute/storage costs.\n", + "\n", + "2. [Make sure that billing is enabled for your project.](https://cloud.google.com/billing/docs/how-to/modify-project)\n", + "\n", + "3. [Enable the Vertex APIs and Compute Engine APIs.](https://console.cloud.google.com/flows/enableapi?apiid=ml.googleapis.com,compute_component)\n", + "\n", + "4. [The Google Cloud SDK](https://cloud.google.com/sdk) is already installed in Google Cloud Notebook.\n", + "\n", + "5. Enter your project ID in the cell below. Then run the cell to make sure the\n", + "Cloud SDK uses the right project for all the commands in this notebook.\n", + "\n", + "**Note**: Jupyter runs lines prefixed with `!` as shell commands, and it interpolates Python variables prefixed with `$` into these commands." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_project_id", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "PROJECT_ID = \"[your-project-id]\" #@param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_project_id", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "if PROJECT_ID == \"\" or PROJECT_ID is None or PROJECT_ID == \"[your-project-id]\":\n", + " # Get your GCP project id from gcloud\n", + " shell_output = !gcloud config list --format 'value(core.project)' 2>/dev/null\n", + " PROJECT_ID = shell_output[0]\n", + " print(\"Project ID:\", PROJECT_ID)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "set_gcloud_project_id", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! gcloud config set project $PROJECT_ID" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "region", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "#### Region\n", + "\n", + "You can also change the `REGION` variable, which is used for operations\n", + "throughout the rest of this notebook. Below are regions supported for Vertex. We recommend that you choose the region closest to you.\n", + "\n", + "- Americas: `us-central1`\n", + "- Europe: `europe-west4`\n", + "- Asia Pacific: `asia-east1`\n", + "\n", + "You may not use a multi-regional bucket for training with Vertex. Not all regions provide support for all Vertex services. For the latest support per region, see the [Vertex locations documentation](https://cloud.google.com/ai-platform-unified/docs/general/locations)" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "region", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "REGION = 'us-central1' #@param {type: \"string\"}" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "timestamp", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "#### Timestamp\n", + "\n", + "If you are in a live tutorial session, you might be using a shared test account or project. To avoid name collisions between users on resources created, you create a timestamp for each instance session, and append onto the name of resources which will be created in this tutorial." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "timestamp", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "from datetime import datetime\n", + "\n", + "TIMESTAMP = datetime.now().strftime(\"%Y%m%d%H%M%S\")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "gcp_authenticate", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Authenticate your Google Cloud account\n", + "\n", + "**If you are using Google Cloud Notebook**, your environment is already authenticated. Skip this step.\n", + "\n", + "**If you are using Colab**, run the cell below and follow the instructions when prompted to authenticate your account via oAuth.\n", + "\n", + "**Otherwise**, follow these steps:\n", + "\n", + "In the Cloud Console, go to the [Create service account key](https://console.cloud.google.com/apis/credentials/serviceaccountkey) page.\n", + "\n", + "**Click Create service account**.\n", + "\n", + "In the **Service account name** field, enter a name, and click **Create**.\n", + "\n", + "In the **Grant this service account access to project** section, click the Role drop-down list. Type \"Vertex\" into the filter box, and select **Vertex Administrator**. Type \"Storage Object Admin\" into the filter box, and select **Storage Object Admin**.\n", + "\n", + "Click Create. A JSON file that contains your key downloads to your local environment.\n", + "\n", + "Enter the path to your service account key as the GOOGLE_APPLICATION_CREDENTIALS variable in the cell below and run the cell." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "gcp_authenticate", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "# If you are running this notebook in Colab, run this cell and follow the\n", + "# instructions to authenticate your GCP account. This provides access to your\n", + "# Cloud Storage bucket and lets you submit training jobs and prediction\n", + "# requests.\n", + "\n", + "# If on Google Cloud Notebook, then don't execute this code\n", + "if not os.path.exists(\"/opt/deeplearning/metadata/env_version\"):\n", + " if \"google.colab\" in sys.modules:\n", + " from google.colab import auth as google_auth\n", + "\n", + " google_auth.authenticate_user()\n", + "\n", + " # If you are running this notebook locally, replace the string below with the\n", + " # path to your service account key and run this cell to authenticate your GCP\n", + " # account.\n", + " elif not os.getenv(\"IS_TESTING\"):\n", + " %env GOOGLE_APPLICATION_CREDENTIALS ''" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "bucket:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Create a Cloud Storage bucket\n", + "\n", + "**The following steps are required, regardless of your notebook environment.**\n", + "\n", + "When you initialize the Vertex SDK for Python, you specify a Cloud Storage staging bucket. The staging bucket is where all the data associated with your dataset and model resources are retained across sessions.\n", + "\n", + "Set the name of your Cloud Storage bucket below. Bucket names must be globally unique across all Google Cloud projects, including those outside of your organization." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "BUCKET_NAME = \"gs://[your-bucket-name]\" #@param {type:\"string\"}" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "autoset_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "if BUCKET_NAME == \"\" or BUCKET_NAME is None or BUCKET_NAME == \"gs://[your-bucket-name]\":\n", + " BUCKET_NAME = \"gs://\" + PROJECT_ID + \"aip-\" + TIMESTAMP" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "**Only if your bucket doesn't already exist**: Run the following cell to create your Cloud Storage bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! gsutil mb -l $REGION $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "validate_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "Finally, validate access to your Cloud Storage bucket by examining its contents:" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "validate_bucket", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "! gsutil ls -al $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "setup_vars", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "### Set up variables\n", + "\n", + "Next, set up some variables used throughout the tutorial.\n", + "### Import libraries and define constants" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_aip:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "import google.cloud.aiplatform as aip" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "init_aip:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "source": [ + "## Initialize Vertex SDK\n", + "\n", + "Initialize the Vertex SDK for your project and corresponding bucket." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "init_aip:mbsdk", + "repo": "snippets_housekeeping.ipynb" + }, + "outputs": [], + "source": [ + "aip.init(project=PROJECT_ID, staging_bucket=BUCKET_NAME)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "tutorial_start:automl", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "# Tutorial\n", + "\n", + "Now you are ready to start creating your own AutoML text sentiment analysis model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_dataset", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "## Create a Dataset Resource\n", + "\n", + "First, you create an text Dataset resource for the Crowdflower Claritin-Twitter dataset." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_preparation:text,u_dataset", + "repo": "snippets_common.ipynb" + }, + "source": [ + "### Data preparation\n", + "\n", + "The Vertex `Dataset` resource for text has a couple of requirements for your text data.\n", + "\n", + "- Text examples must be stored in a CSV or JSONL file." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "data_import_format:tst,u_dataset,csv", + "repo": "snippets_common.ipynb" + }, + "source": [ + "#### CSV\n", + "\n", + "For text sentiment analysis, the CSV file has a few requirements:\n", + "\n", + "- No heading.\n", + "- First column is the text example or Cloud Storage path to text file.\n", + "- Second column the label (i.e., sentiment).\n", + "- Third column is the maximum sentiment value. For example, if the range is 0 to 3, then the maximum value is 3." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "import_file:u_dataset,csv", + "repo": "snippets_common.ipynb" + }, + "source": [ + "#### Location of Cloud Storage training data.\n", + "\n", + "Now set the variable `IMPORT_FILE` to the location of the CSV index file in Cloud Storage." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "import_file:claritin,csv,tst", + "repo": "snippets_common.ipynb" + }, + "outputs": [], + "source": [ + "IMPORT_FILE = 'gs://cloud-samples-data/language/claritin.csv'\n", + "SENTIMENT_MAX = 4" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "quick_peek:csv", + "repo": "snippets_common.ipynb" + }, + "source": [ + "#### Quick peek at your data\n", + "\n", + "You will use a version of the Crowdflower Claritin-Twitter dataset that is stored in a public Cloud Storage bucket, using a CSV index file.\n", + "\n", + "Start by doing a quick peek at the data. You count the number of examples by counting the number of rows in the CSV index file (`wc -l`) and then peek at the first few rows." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "quick_peek:csv", + "repo": "snippets_common.ipynb" + }, + "outputs": [], + "source": [ + "if 'IMPORT_FILES' in globals():\n", + " FILE = IMPORT_FILES[0]\n", + "else:\n", + " FILE = IMPORT_FILE\n", + "\n", + "count = ! gsutil cat $FILE | wc -l\n", + "print(\"Number of Examples\", int(count[0]))\n", + "\n", + "print(\"First 10 rows\")\n", + "! gsutil cat $FILE | head" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_dataset:text,tst", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Create the Dataset\n", + "\n", + "Next, create the `Dataset` resource using the `create()` method for the `TextDataset` class, which takes the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the `Dataset` resource.\n", + "- `gcs_source`: A list of one or more dataset index file to import the data items into the `Dataset` resource.\n", + "- `import_schema_uri`: The data labeling schema for the data items.\n", + "\n", + "This operation may take several minutes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_dataset:text,tst", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "dataset = aip.TextDataset.create(\n", + " display_name=\"Crowdflower Claritin-Twitter\" + \"_\" + TIMESTAMP,\n", + " gcs_source=[IMPORT_FILE],\n", + " import_schema_uri=aip.schema.dataset.ioformat.text.sentiment,\n", + ")\n", + "\n", + "print(dataset.resource_name)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "train_automl_model", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "## Train the model\n", + "\n", + "Now train an AutoML text sentiment analysis model using your Vertex `Dataset` resource. To train the model, do the following steps:\n", + "\n", + "1. Create an Vertex training pipeline for the `Dataset` resource.\n", + "2. Execute the pipeline to start the training." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "create_automl_pipeline:text,tst", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Create and run training pipeline\n", + "\n", + "To train an AutoML text sentiment analysis model, you perform two steps: 1) create a training pipeline, and 2) run the pipeline.\n", + "\n", + "#### Create training pipeline\n", + "\n", + "An AutoML training pipeline is created with the `AutoMLTextTrainingJob` class, with the following parameters:\n", + "\n", + "- `display_name`: The human readable name for the `TrainingJob` resource.\n", + "- `prediction_type`: The type task to train the model for.\n", + " - `classification`: A text classification model.\n", + " - `sentiment`: A text sentiment analysis model.\n", + " - `extraction`: A text entity extraction model.\n", + "- `sentiment_max`: The maximum sentiment value.\n", + "\n", + "The instantiated object is the DAG for the training job." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "create_automl_pipeline:text,tst", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "dag = aip.AutoMLTextTrainingJob(\n", + " display_name=\"claritin_\" + TIMESTAMP,\n", + " prediction_type=\"sentiment\",\n", + " sentiment_max=SENTIMENT_MAX\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "run_automl_pipeline:text", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "#### Run the training pipeline\n", + "\n", + "Next, you run the DAG to start the training job by invoking the method `run()`, with the following parameters:\n", + "\n", + "- `dataset`: The `Dataset` resource to train the model.\n", + "- `model_display_name`: The human readable name for the trained model.\n", + "- `training_fraction_split`: The percentage of the dataset to use for training.\n", + "- `validation_fraction_split`: The percentage of the dataset to use for validation.\n", + "- `test_fraction_split`: The percentage of the dataset to use for test (holdout data).\n", + "\n", + "The `run` method when completed returns the `Model` resource.\n", + "\n", + "The execution of the training pipeline will take upto 20 minutes." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "run_automl_pipeline:text", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "model = dag.run(\n", + " dataset=dataset,\n", + " model_display_name=\"claritin_\" + TIMESTAMP,\n", + " training_fraction_split=0.8,\n", + " validation_fraction_split=0.1,\n", + " test_fraction_split=0.1\n", + ")" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "deploy_model:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "## Deploy the model\n", + "\n", + "Next, deploy your `Model` resource to an `Endpoint` resource for online prediction. To deploy the `Model` resource, you invoke the `deploy()` method. This call will create an `Endpoint` resource automatically.\n", + "\n", + "The method returns the created `Endpoint` resource." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "deploy_model:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "endpoint = model.deploy()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "make_prediction", + "repo": "snippets_common.ipynb" + }, + "source": [ + "## Make a online prediction request\n", + "\n", + "Now do a online prediction to your deployed model." + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "get_test_item", + "repo": "snippets_automl.ipynb" + }, + "source": [ + "### Get test item\n", + "\n", + "You will use an arbitrary example out of the dataset as a test item. Don't be concerned that the example was likely used in training the model -- we just want to demonstrate how to make a prediction." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "get_test_item:automl,tst,csv", + "repo": "snippets_automl.ipynb" + }, + "outputs": [], + "source": [ + "test_item = ! gsutil cat $IMPORT_FILE | head -n1\n", + "if len(test_item[0]) == 3:\n", + " _, test_item, test_label, max = str(test_item[0]).split(',')\n", + "else:\n", + " test_item, test_label, max = str(test_item[0]).split(',')\n", + "\n", + "print(test_item, test_label)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "predict_request:mbsdk,text,tst", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "### Make the prediction\n", + "\n", + "Now that your `Model` resource is deployed to an `Endpoint` resource, one can do online predictions by sending prediction requests to the `Endpoint` resource.\n", + "\n", + "#### Request\n", + "\n", + "The format of each instance is:\n", + "\n", + " { 'content': { [text_string] } }\n", + "\n", + "Since the `predict()` method can take multiple items (instances), send your single test item as a list of one test item.\n", + "\n", + "#### Response\n", + "\n", + "The response from the `predict()` call is a Python dictionary with the following entries:\n", + "\n", + "- `ids`: The internal assigned unique identifiers for each prediction request.\n", + "- `sentiment`: The predicted sentiment value.\n", + "- `deployed_model_id`: The Vertex identifier for the deployed `Model` resource which did the predictions." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "predict_request:mbsdk,text,tst", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "instances_list = [{\"content\": test_item}]\n", + "\n", + "prediction = endpoint.predict(instances_list)\n", + "print(prediction)" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "undeploy_model:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "## Undeploy the model\n", + "\n", + "When you are done doing predictions, you undeploy the `Model` resource from the `Endpoint` resouce. This deprovisions all compute resources and ends billing for the deployed model." + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "undeploy_model:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "endpoint.undeploy_all()" + ] + }, + { + "cell_type": "markdown", + "metadata": { + "id": "cleanup:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "source": [ + "# Cleaning up\n", + "\n", + "To clean up all GCP resources used in this project, you can [delete the GCP\n", + "project](https://cloud.google.com/resource-manager/docs/creating-managing-projects#shutting_down_projects) you used for the tutorial.\n", + "\n", + "Otherwise, you can delete the individual resources you created in this tutorial:\n", + "\n", + "- Dataset\n", + "- Pipeline\n", + "- Model\n", + "- Endpoint\n", + "- Batch Job\n", + "- Custom Job\n", + "- Hyperparameter Tuning Job\n", + "- Cloud Storage Bucket" + ] + }, + { + "cell_type": "code", + "execution_count": null, + "metadata": { + "id": "cleanup:mbsdk", + "repo": "snippets_mbsdk.ipynb" + }, + "outputs": [], + "source": [ + "delete_dataset = True\n", + "delete_pipeline = True\n", + "delete_model = True\n", + "delete_endpoint = True\n", + "delete_batchjob = True\n", + "delete_customjob = True\n", + "delete_hptjob = True\n", + "delete_bucket = True\n", + "\n", + "\n", + "# Delete the dataset using the Vertex dataset object\n", + "try:\n", + " if delete_dataset and 'dataset' in globals():\n", + " dataset.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the model using the Vertex model object\n", + "try:\n", + " if delete_model and 'model' in globals():\n", + " model.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the endpoint using the Vertex endpoint object\n", + "try:\n", + " if delete_endpoint and 'model' in globals():\n", + " endpoint.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "# Delete the batch prediction job using the Vertex batch prediction object\n", + "try:\n", + " if delete_batchjob and 'model' in globals():\n", + " batch_predict_job.delete()\n", + "except Exception as e:\n", + " print(e)\n", + "\n", + "if delete_bucket and 'BUCKET_NAME' in globals():\n", + " ! gsutil rm -r $BUCKET_NAME" + ] + }, + { + "cell_type": "markdown", + "metadata": {}, + "source": [] + } + ], + "metadata": { + "kernelspec": { + "display_name": "Python 3", + "language": "python", + "name": "python3" + }, + "language_info": { + "codemirror_mode": { + "name": "ipython", + "version": 3 + }, + "file_extension": ".py", + "mimetype": "text/x-python", + "name": "python", + "nbconvert_exporter": "python", + "pygments_lexer": "ipython3", + "version": "3.7.6" + }, + "vars": { + "AIP": "AI Platform", + "AUTOML": "AutoML", + "AUTOML_URL": "https://cloud.google.com/vertex-ai/docs/start/automl-users", + "BATCH_OR_ONLINE": "online", + "BQ": "BigQuery", + "COLAB": "Colab", + "DATASET_ALIAS": "claritin", + "DATASET_NAME": "Crowdflower Claritin-Twitter", + "DATA_TYPE": "text", + "DOCKER": "Docker", + "EDGE": "Edge", + "EMAIL": "cdpe@gmail.com", + "GAPIC": "client library", + "GAPICP": "SDK", + "GCP": "Google Cloud", + "GCS": "Cloud Storage", + "GNOTEBOOK": "Google Cloud Notebook", + "IMPORT_FORMAT": "csv", + "MILLIHOURS": "20000", + "MODEL_TYPE": "text sentiment analysis", + "NOTEBOOK": "sdk_automl_text_sentiment_analysis_online.ipynb", + "REGION": "us-central1", + "REPO": "ai-platform-unified/notebooks/community/gapic/custom", + "TENSORFLOW": "TensorFlow", + "TFLite": "TFLite", + "TFServing": "TF Serving", + "TFV": "2-1", + "TITLE": "AutoML training text sentiment analysis model for online prediction", + "TRAINING": "training", + "TRAINING_TIME": "20", + "TRAINING_WAIT": "60", + "YEAR": "2021", + "null": "null", + "uCAIP": "Vertex" + } + }, + "nbformat": 4, + "nbformat_minor": 4 +}

8P2{ukTMbg$eLN9*LtM4fM)YcZXUKvK)Gy}mgDJG%p0P&-CE0zS^TKXn8s%FX zj>Wpw2m&%jDXMtaTQ^}echqj2pkacEtURp&c<%)+i({Y z#0eiTHMbr;>x@`e3E;;iA7C(=OKxDfLc=c;AkmPAwu|3In3IF2S;>{fsY63DZ){{G zK%Vc)3-;GYM>1r#5`R|iufd|Ne7%kx5wQm#Yl6urqZ;|AvTcmT-qlE>RA*PE0Zm-JONwK}!RNuel5&OE5!L!x>&iue?@|JfO7 zNFIpOGq^%P4jzH6<>5V(s3W3=UEj!R#=yVeCjq4kmQWG+&4;rbAr-94a1h{cN{PGb zey2e7wYd3FZvX+&Cx{%)kqc{t+7A4kzCqOL#2JrqJmCOHBLa?T|1ME8bdbdK?G;v* zs4EGsrhDH`=F2_qMoq-Ij6x1~1BAJIYCo3*VVE*V*!J|pDQvra>H4h8v$Cv*)0&Xn zQFN3pHHWxg_3XF=mQo<~=y5PF4Q4?FwP6K%0?n{Y(J{ag<|9)I$5Z zM;KB8{sCXZ+PY#ZOaq=mS+W(6!UbBu=>tR8J*$R(5%@*WMJNv!k?it@mRs>U0ldpJ zfKTKgv0ssv$Jj2#D)2LYpzK_BwDRrY_`Hr6EuZsQo31>O;%03DY4gc37z%Qz-m(ut zZp1!5r+d6ivG5wB`Ees}p!F77)i7ClIOT_>Uma@pTA+%oD2g=y!1iv7+qrg%1vj!B z$Wz96(#-DT6UF`YGJg&=XYU1u*n9UO;k`?GWd5pGSO3aU7Wnh7@0oZ^a_4qzo6VlK z^NLG}`na6>!+uICFj=$zkY7*|q^DO#7GK=-p^hB4H2jB6R! z8Cdfy|FXU0WiG+(<1qPgDg`@&``n2${r;=&WU3x#9D8Q3( z8KEyNmx>ceTs*rHdFlPr3m^}fa31f^I5!-mJP(_KW=VOOvQwRIJ2%#8n}%}dPAK7M zBYN2RCak_XehOt0KIK(KZf^sf{C>30Z!bpFR0u&CSJ~6fM~p-gTd{vanM&@+3(Dw? zk^GR+Q_J(8`=a9AF(JeVz!iM~9Z--xqgE(-_Y3aG^KUG?xo@(&s54PhBb_c~$}&jO zv;V?H?Kk(c+Z~jlyhG^pBVGAR{>}IZ6n$x4@NIr=ZONzYk&TXCIAUw#)EFtuy$-Qe zOr#_~8QiiiG6|SfGUBd!rwp0z?3<8pEgHtVTTH5=lwWFE5oT9&xFMruRyd~=ylZI6KcvUsrP<;H7QQP0Q3RkJ~0nCjnow}~#m=&}ZMbVY}kntc@jv<73UCZ}5 zf%bJPSI#Ms`c2OrF^)0GqnD2zex3>To(yiRxZCXo3B< z+}|P87G}@6V5zb7bdTdQvaGTu&~PM}^=Hg;v%ZSS3f)sF95h;^Ei=s7yZnn0xVkXL zk5=doSzdR`Ib$?Bl*d9xLrBy-y!`-;74onwr zX?zRRU3{>^9+x*ZwDCQOGH{vK<%BR2 z*I8<#zcyus0^A9BiTBhD9G;plsK#|1d#*Wt{wsmh3(LoSH@!f;#7`7&58eB?!Vt`o zJpZiU{#Z;sCa=tuK|)qRis?9XR>f(F15wR30U2#n#U=!BY-nAL^m3lQi^>sLO!CQ5-5@k+NP%uUthVm(=}BTo9UWbu`zu&$$g#W zMIyb-t=8wnfx$X~k(X`L=&F()-#F?Tv^9IR?Tp@c&%+W8X8My4nJ4+vUqUVesj+HD%=QNV$Rc=;# znU#R>*?g$K8Xm!SydU4_`boj20&f5^l7%h(8_nuhr5;cmsGaNC*>fx~KnM@X_K_8dd+Z3qM@Q;? z1DkuF`gH!YlIQ{4=d=hF;*O?U;lf0L5~uz49&oe0!2nun8pTit+T;5O7kHm4?H>Ab zm^=AO0PA$0KnkxtGdz88sQP}U)MFn`UHMlqK>!2Qq}?kzbn(eJd1|!@`&D%P?H; zT)n5m#jzSlE!2|$aqVrY#%-=d@;%{8m|Am;Eqip*6?-hRm{c;lmsP$#0C3IgeNF)E zoq7f{&*-*hDz3gxP{(?hOS0ZRxceGLxB0IGeH}*=-ZTzw9UPBoSyO3AtU!rW`EwWC zDD=1Hxt@TEAiir+{b~yIl_ltHOI8$bwr`1?iZzdf<9Hzya$D+iuT4v!SYJgR4oZ{e z*)&u-WAESKi*=kd-dq_IT~y@e50V>fC!yNyC>KE`2>Hvd5UXkwZW*hpftS9 z%KtO2GT|P0XjSQx3>tq`LsoU+(?|ApJQTFInY!_j7X+*{956igXjNT+fJM9qf7VZ) zqe5T3ptyi)VuKygU}Zf9)$>m1TF}zXP8#HwWI=_mIXsfNyuu^<9@ipItElMjc0}C^20AAFi zB!~%vwFQ4I7e`ND1&Ri~4fyV4nTnTBe+KTcnh1k0?1XEAuM%W>O%}d|sCfcLp`vRH ztI;-e*09y#pum-XZrA6xyTkEhd|vLrZ{@!YtEcF9Nj|>c z`xwpkH!m#DI#4I^{nyOlx9+US7?qefre>*2&F!)Q&0Bm=4T^{f8Vf(PMH=nmSo-A8 zV!ptu{tf`H;_1;Lx5>hU)L_b-GW0m!U(puE@80t7C)Q~}LmBQ8(^zSKb0a-DxF-mr z3@$)iXbub_ND?()clD#Q;{qa#Z=fr=2-J5cRzMNzNH9T93)R9P!@_6R?ryB#QD4!+ zV`Ym_ZVaO$5N5*SL_?RWpXdJe@BPF3)cG-1DJ78q42plF$aTyvBUD>kjJtcdGTYXV zCbQ?!9hJ5qu*a==3knS4N0bH*h?Ddv^E!8)1zC4ej&B>p!&YU&!yg&}`}z)Yud870 zq-1@f%GC_mEedHL^V1U-?jyL5>5a|S-+T|zkBkXW^^whYk^yjx8^rR5 zQ`@|qrj`$c=r#R`qgGA}3X316j^glf#8Px8^w|QMQ?Nf~iPBZ3BLi6-sUh5v zp12^sFs=Yib!(=GI0~k8NCXe(mrtz~A1a&!^i6Fzi;^$>UlZo}hs9P!mq@h9`V?3V zncQcOQ!1SSJLGkAs>HQBgvb9pKGW%6OyIDZ)4Ni?+ZecoAKrE%CffB{JjWr9J(OD5 zT0pe#t0>MGE7A+D zytY57N;=%Y3vH!s!XXdA`vS^c%_1t{JvK}MciAocsgW7JJO|WVp20m z&$2XxUx-!9&8|Y1u}=FQS%NKpkxOIQvU$!GaTC>-Fi8h;Jwrg~(8juDQUym{tILkt zj!!gj=#{7~hCnX_zdk5@odQo8DJKW$OEoH=ay;o8SpIHh6RQ6;r2y^kNLpw4Ob z^;6M4RDh=X8i=hyU%5y2RG1a#{{Luok@Nja6 z2)%T2Klb18RKR(_NvOs;Z1{6Nm2&p?%#bcjwgmrx;J3lY**3M`%C-ca6i%hKT{%uI zmy`FKpt-}c7p2YPK@Wx2kZhs#d;4hZQJZuM+=FPu82(y99LG;Zf{?hNZjSVs+@~Ne zI3Zp@`{T(X7eH$AUvE=Fk18$W+LJ0!??`wd2+EWlm!LUCHmQvfo<%9MaBoi;pO)lf zJ8_KCt(4Z5%|-!se}l=(chw_?VcqZOI#E1|d;Ad>^d~H+J?gk^voJYYURP3Af=eTO zL)rht!4RQ*7YD!K z?TMlb!td#Ld>ZA`s0e9Z0ce*@q~K??nSQRH!=jK!gMdsl3q8| zXwGN7bBE2U+CP05`15q@(9`6S)1An09+}9IGl1JvFNpYon)h&CI>-Cp3>fOZ#M`hv zkwFf)j`sdq>zcJ{Go=K)>)T8G1d`vaZ&2R{eT0Kp=8b-z`r|`_$ zCb~E!hn(K;%*ramAzLjQfMFC7`#drs9$N-QTx6_lv|Wb1nIB)x1q@NN(89rVH!=wEc)AJ_03xYJDP&Wj9J*6B-Tm{_Q3YknA~w)a`B z?R|kU-ZMJnAV!Hq3yvyn#^?+GIfh2J3c!=hg}RiFNg@YC&6VLETlf{V<|D5SKe|K- z-#=0N>kd;Hg1c<#DGD0AL)0r#mO{8 z&%Qe#s=^7s@9&&WKRGFLt5yW9Zc$iMI!z0c&x;DiGfvjxu`TPAW9AvlDqG=x^@2tA z2wBoz&LsFd4C#C%B^OC%KB6J+qu?Qjt)D~+RZ6qYCX!R)UKkB`gfYAXCsRj0`$_2f zL7*n`4D*Ok#9!c~ zeQdVh7BQ*DK|x5+9%qQ(>m)vV{qyiw==CI|C$y;;l-}dYeZK_FDUk_+nDHV0Y#$9Q zE*s1P&x#8F?fVzQUtM}XUD5k+@d#3r0_YJg@lZJRFUe%zJ$sxRl2VxIGF;1qaQm`0|%K$LG;M9(xE@_b-_)z%ZH+b_0C3bJ`MXE7b!jY6jtQd#{{4t2okz5i3gsAnmjv&s*J4DAvRl^D z>{_slBg!^aKjVt07Z;((O`NNLZI+{=Y9HA-I}9>QX8VrUtVf3TxI@xZG}P#|hSP(e z!NXW{>XB9k=GUR>zsIt%y?`P+E*lkb9tVL2ZkG+13xb}~gZ@C93THX?3j+9uKvV!7 zshrc7Xx|fpST&(lAJJ&B3y$Y6KfK!d70MQV?N+shsebkr$;) zU7&R?bb#kg&)Ao9#U%!xTJ?KB?s(o)xJROA9`xdT zJO3Mj?D3nPLt>4c{Di;&B8C1cn=25ob8Uq?Om(AdUvygy?>))fa|St`TVXZ0$a{x3 z8M!4})UAE*ev4BOt@r#l+r8oYk>{^=$K- zGWEsBp%jiT;?tz`G(Drgt#AwW5+o#)A=Ji zU7dUQ%N*a52d9k+PdB7?nB#j^Eby1!FJ-p-d2itJgg85V}6x^D$@nrKR=nCVzT&~!>UIt+OkhItp`AhAll*q7} z$b&c4d6WrR$m^B5O7^~LBCGzlC*6BrAf_)>afoM=s~cL;A?m6iLM0wu^}u%0&EmWp zNJJiEYT7XwP&N#rK47;!tB$KIsr^u7z74 zs#vxVT%{L?9l!gPV$e-|)J;UC-OV5MLONGS7!`S*BMSv-w2jSc_&qvPT%KBCJ0o+x z(F5-V^_?ycBG6Vc;%>ytoSDYejZqTh!rVS9y#jKK9zC?o@zC4O9YCYaMbnxKTd%Zq zIyM|#j&2gALG4$-*y~D7guJ`WB9fOy1YhzE?^nG(`D-b7X==pW=QJ{ixn6ZIoN=d% zSy`pw=%+3bC=Tjs`@S4Gl{-7io9cKu53`sv9C(}4&1caU$9U2QZr~9kN;$S6VrF5( zhY4%^-&|CpBtq^CXs(p6XHWeMfMOpzc?4#&iU`CU5$n-B`v}qM7`ITAmumyut6y? z77Hf#ENHYbkg^RevusO8T3o(BOpL0NA54Uzqm~QqS~(qu^{CwsgtM6F2h{Bm;@Q(I z3xP^Vi~77qhg^Jo3U8+>ySr}**eQtsen;W=EGp;#%+Tegj)2c*zf%jLTb|Owf+0S@ zZrPI`&hJKNeF5Bs3_}cTB@|93Amoy`UOH$Jx>qYU`88bY?Gr98H1S6n$y(1Cb!W}- zAJK~DqU+ovMph?Fn*KbE&(x(J_~^iP!BLn`80y$xJDrlD)&y`&yyoJ_wvW^N%PQv? zKRt_zp=mqMO85KtL=ReXbqHUA6D*q29Q3F-YfGTBntAGqhHZtqGBwIWDr9Dl^xxy#~%% zLX6pWKlgi{aHd|c)B9nHX!37s>3!Ku&}h)EeN`1=GMo6;KmW3eSyS&9P(cNQ*;;f( z_`dIeq)-=EB+iz|_fviVe%d^^q&(66{zj*0edaJBR6pEhAbx^84lL^W|R#noM(DPj!D+@O@T5(oCf(`S|-+3L?-h#7j>C z7e{@97n8Suq-FgaHUrfK#?Th-55oAPxs+{UEU_Cr4k-_?9}zq(yW9B)-W_mw zS1i%W=M=Z}X@T@YS{f*f8uLaosIcGyFw1k68 zPZI00OLg7v%+Ry1vy$wpbVvkzXdKa~_ny!2W$3Fwr+^*D;?1nXY>wLb;zoNv-wVi! zmvHPirUXcb$LL$p{FR(Ii=aSFq zNSUKzRD8Z01-LA^8VXG2iyNoEq0?+wyNW>#q~b+D!Q(PApSX?Z32Q;iUDP%9C6xXW z%_@rdLsX>sbAh6Pmn}H#3_eG6s51?I*rQB3Z&A9M0W~dVJ=BI*_qpL40VRrSicNQK zNT2H$p%J&eR9~k?eW+Pe+6V5RYR}NG$#>q#Kn|FdLZ2q`77dh`tpe`0J1YOyDQQB^ z8DRbm0Bi-R7tR2M6?cB4__}kaji3Dp=cWM16GlimMRG!=b~fPNzMicpWk7#uJ;XqT zeI28w%!sRJcLRoXTcS8msS980KaAi?z^{p6=g_Cg7@-r-eMA_zUh+aUO1@w!{W+>s zww!8=b%-tvFTVcspf1PgM^+Ip+;2AO+`kM1E#e9P#im=|d;beJ;WU?ieiv=)t({*x z2K^s?9tbftCc9|Ved2goh;7%1#X~zq=n*WWX6_Mv_I4#LvS7*LpfAEqPZv~zvsxXx z_D6elXbiPPW|Qo8@?68gD|sM-V#TlUYo)$BC=um?Zb-#LdcyYnGDat=Q^Cf_+-v5J z8(Jr8$Fw@@%dFs>3?gQ&Y=(oOf%Q@Rw=I_<-jcE6(2m?6fNuoc|D86 z5=mArzY@#exhHzV5GdD?o>A5;w|A^;tT8Kx1$f{lB-e3GAym67jn^)lCm!ch+HBJu z9%5(D{bZ?xxUxewFG)=_N>^R=T`9joqCwO7C6NaLNu{9I=r}T}JBve$OCz~Zem`?i zzGO3t^6JGjWt9{${wj0kyhGn4h5?zFHiV8Pw1o>S+TuV&i~ zx_Gn#gv5zoXYW8~1HHGXoe?Q^d-kdOZJkDiq$a^oXQCS~_JYHNh4!zTazet)`9crq zO@r$YJJ}+s;BtXDt{S9FT5nN0n)xcJ>953Q5>^*>NQ;K;q@8WMMvy{Q+)Pbn`|t#i ziM8H#bx=H~4|KA7Ln^_#LkDeMW&-l3eDunE$u7x zig-yFF|!i;*F6|CQ?vtOG5AW7W%7PMY+rPJ*G=4p>|GyIf`vKxK7I2W%oed81{g8l z(yc=oWApf-V+OZ*Alaa(YRor?IrOHlCBE~bCE%b%)oJf964H2mYs~4ei<aXLauFpeC>h(4wl!;!8y8nJ)% zdZO9M=5F#>MgKhon~qBLtE+TNbcU{XiV3^hT&2C`ETAl;I3;u$No!UsbE zmFwxlr)E4sM@_XvD~@4l(5spR*^cnG=Ibiq!m0PrAC*gq?3j+*MBru&uDa!?o!S{m zY!ofwbY^sv&f}2Gw)!bd@*#G&dH7VkyARcIX#^en)uyYTc|ixoUVuypdPi$1?k?wp zxKRphk|x26GkmWklQZfK^Gh6^2OV2GM_C0kB-UjRS4g#=i1~shgAj|0u4&(exz!)H6AvdUB{N`^?`$#J^f=9k zctcV-548o&Mla6dVymiO6Tn0&`0GH{X$t35D8n5T!^ntvAEn4`2*#k6gGK_c^s8qq z3YANAPq>Nhhb81D1rYn>SZMzhuqn<|^i}*pGH-c0=8{V2>=kL^M>yC5O#xUVoy(%p zKhEKGF^VS4r0qkPvI#anT12^N6~IW@3!Y3wLPli2x}H%maeTIOtIq~GR( zS>gOlLVWJIshWOuegi2^U?i5pM8s9ZHBE_3xg)I`Dv3+WqFp68))rYg6((jTFllMh zhfnWUGH;2;yk(1f>v-vkZ1w_1fUEO}-=l-vxo3yvN|`u+eC3$+Y&IQC4D{jWBrL}8t$@B3!S~OhEI(uL zw%Wod>BY)%?-Q5z)`MH|mY#LO=a7@|O;&Rf`c4#rJH629{I-4 zwROM>LPh;R{Mi%zk7RNj_d1Cz@WQt_fIW@OF#`>T%`CHVT~ zFk1phG-?`GNjlx55=O4Lyv_7gDHuRz9wj<_Vqe8`xL#hHvLOsEB&1W28p`edEgjU1+Zx+BOJ_s!C0(Dtt zYUeixmcj}e=I`G@q9r|x>TXJ_P&k!7#c{Va?4g(SLb85YW2pcG({g%zlp6*4gIRPb zlwew9a+r+k*zlZ?hF~-~?0qw5hHut5$fpPl!3=<_7*RD8s~Yh>214HUJly;wcKhJ3 zT$IswX>-^ErD$kJ_Ko&kjFh_w{APpq zas;I|rua$DG&|DC#EhiPqW-j^&nH*Il^l)NUl;5w{0JDv*m<8kRIU8$gxXqF&poni z(7`x%tZ^cYO(&e-mY2g4)O9%fx`@5Wj#f7ECh_@Q)H!Hs;2B40O59GEuhZE*e{Hfe zV(Veec5Fox#G@ehKBs{bx(>7;`LGJt75C#eai&&BWpvH_BCJkiu%r+ywrq|*KmM_C z1?>XPo#{y*JSVO}^y)D_T&k}seQZcNOWWEX`z*A-jIl!D;tXWPpnxJ-!uWjF{oIV&^1dmY^B@jtrz!xXX9!r3J`~w#9 zZNB%^XXeRgGr7=tSgyzI`KDg?;2h<+HrpGz!Lkn&k9Y8l27g*Nu_bOMzKV2z92%Th zzgWArW9KK4C*~~P*n=&Axk7hAt?aa#u=?q6O;O#?CS)R+-(f$`>FQVaM7C0j-Y}@+ zkf(^P84e;3P7ROctSo1S>)?jQ<2xE$$dtkmd( zj4%~V{O2LDqSD5o%V2EXS_nu5lU0Vo0L_OsjOh_7`F;HzV44Zzk#=NpbVbn&!q-G} zia)rhPDGlDV1F}By;?usJ+ShXJEP->c4m@X^GVQKx7(;H%CXW-kPtQ%jJw)IbuXZ*AYXX7Fr_ z{CGEtwqc_lzRv-6ky^}S0zAbRqlwj><$HB#1(-fdQf0t3#Hk#R(?}g+D49L?L;HT1 z@lYJr2CDcmNtx^eUqueZrp(K14uDa?Pf?6DBhSaDcLpA*LulUX2A_Inw(+Go{+*=~ z#YbP1-9jScECFP#F!oT6L?l0Qo$+AAv^qnYVv(Ku8w*p?_IfZYf=2mayt!OU z#uMfw5>7>6F)E$_rXv&O!Y6z?1V~b5PPM#pA`IoYek1jBRKfFQ#kj6*PA5?UccYZr z(G)8Y%-ULZ#7`_(*5jWS03+C3ZqlnD60PC0@51&CO(bQH_^SPKer0Acd5B;lJD9O7 z8IE!W&|-UZSJ0Ap+xtm7Q%6KMj@5EtZaZ!4^MS{+N7aNz6gGLN0S$N6no^_jYNaNy z=R@ozmU(?`wDtIz+t1()BgjxZ9SmnM77y!rGQL?rFO{?|rQ}#z8B)zc&qYEMO~)1* zJZKB~1gbfdG}w^F0+oGO;!(FU6C$GOiU~CIyGl#Sv4SdLN>j$P!m{8g51=aFDe8a3W zImkGu6iuNtz*IOC!4()1Xx{QJ*ht4l8k)|u=KAylpU#$!5ML70CL_*@<2%clc;R6TGu!{3bS^20sTw z_Hva8jgvwVd}43JJCPxEk4-HiW!skpJNw&~C?p8_4{s7pO$)L%30BmT1rpQ^0s8YX z`D9n0ZS5Tf@%8MqgEFeqd%&?F$ zR1S`TH3{CiCHKp!HG;=YcuTxW%#iA1aoBejk2^|-nf0%RC5gJnX8WD_*aw5I!W(j4 z`0r>w5MYeXMzcLE6nx86HPyi5bUlC1BfyhM&$98PLN}>)3j{yYc&VFKj2=#HtsByj z^6sZ!bS{}4LIfIc%NbI9vf(wqH{2QN`g;^F!SR9wdIu>l> zn__q+%XPsYCt#k_C(hW1F?S~?cF{22FZ%HzoLnjLBU+i_txh}k=iQej@_=xe{W#n#f zR`KhBdgecK!*a$VC7wdA#e|e=KqCoM@X52LLtO&C$;2nxU#SG-44h}2CV)UJ@$Cg8 z03cJPdz;;n^%gSw*XKnPF@p)wAeQ}IM=xh$6AX&tO5ykubn$vokO`TS#CMF|GXwYCGwp@8|71k@a_|kO7B&j2&j2~ zKy!>J>aa4B*#g$qi3!WJr(gHa8D9X?WA}1z(pWzu1z;$F74x^BMo2&tU_3@a&4vkZ zkR*KpuYr9+-yVGekGt(9jExp^m5P61s@iV8C1LyRhz_iiqQdtKGH;;+odGf@U)0PBDFv(RL+YRR8`T@L#`~-XEN> z(XBIoLHwcr{NP9Q@OkX63xYz?KMfBNA@0pNv_(dY%o{gWSl z1HByn=`a5Pra|RNjSpT#fZdb7{`gsgb=XOr?!CBZR|R77A1t>2`nP_vAYM?ne0%Xv ze)xasi|*mO)hQMHKpjLmN!kXudp4X3E}%%J_|Wrt z2rDLl3%>S9%23M$*1}$%*fE#m2}?`Bx@crNDMwAQyE+;R?{Gb-o9zW;ztrN673boL zA^C}6;LN|w#hSH3K#G}hgLccAUkHzzi`jCh@q5v!;vO{v3?~!N_(i8hfe#!47vX@P z@_7fsw)DOrU3m?BFB9IBmvdNK>y`5W!r?x01JS81Kpt&fCDE~k_yr*lCmxAGU#jXly?XpmrejJ`6pYT zFCi)9cZ`N0B71u44ZvB8c=sNFVS1oTcM)Iij^^EqPaK(dgz7??(@yGFjtVTMuOX%J z-h>h2cZ+Wj&4x)PU}#Ipl3Pj_?=C-t*JK739u|Q(41G{hWjs97QfN(1r7LwmChxFf z>9~++X=;2~zmM;fsPR;)0+oZJ{kR_ZC8Vy;1Zv%OoP}m$p6igVAD?-bJw476tXw-7 z#DhFuFC5a78HDDs2Dh8O>j2D8*Rz;c63$CIFZ+`Db1n`@Z?!!#TuyH=y|7ST0v~>z z9#laFZ5%g()sA1!HQ&t$7jEiR0Z+PGd=up8-|S*43AqDdL~yV8KYV+TC+;d5F;>6K zJ&{Fm^qJ576g&yyqe>u&EYZSKvw&D4$D0s3qhglht||!N`A{gto6TBSY**~3>ddC> zGgMriAMm4I0=Beu#lzTnOOAD5B@YUVY`Y>BOTCz0v4HG4=c$h#$1Uxw$szeqK$Fnz zRsRorZyFBu-~WF{vP5&E~0J?~w|b=|AuFlKz_y}Xv^^YMEB>H=tscsP%k1huL5fVm6} zu9{k63vm>A#cMBP$c6zd2<`s2g?;LM(z9amIjrz(>#J*vP_KR$%tCc-c2!4tZ-yBa z7t@{J;tYc6hV?RA%YJ(LN;}SweH_9hsd*lpK}EvYOGOiWd_Vzy~ko zK0|KARH@BK_t)ulW_e}p!u7nMITc&rGgY>zn$N`ty!e{C3K~CgP=6?AQW3JM&ek;a z3~aviK770j`I=K?yuB#e;VT6D3G+wYQpxRaRFTwD1y^MEzfI*lN(+SC2zU+n6bsy z%`%h(YbMY+2VP=|1q>ZQi{()~yT1bhyFyI1+Z}wU)!+rVEKku^Tw=#NLqp%wGs+J& zUb^Xa`0jabxEn3MS_8(gF5b#DaWZ=eFsj5&{LikH`j_*>$5!8pzqLTgj5-IPZni=<<)5p(N@wvK4SI&9RZ0d&i2I=8b0Jl{JA()UtF!w9= z{cMD(=lD`%#6-> zX5uYMqMDr&iJlEkl8aHzgv3Rf#M;Dk$`}DTA)P!*QJdclZGPw4wX#(^mF_%H!nIQ? z_BgcU@@O|vxsSADbH2a!LY4U~RP&u<_Z9@+5QCn3pcHAzyIYA&eG& zfO{htQ4`|KIhIj@w0x2vx7A9g)dLZ;k}mm}K|=SxtmpZox6%C8KB-Q$sf~m2{xSw2SH%tL4W>b=Slc8$%OH9KAS$O~7OCyS#eVC5XKP z?F|Z7_1Uv()rLaqN&ore#svd?1ksVu=dj;5-oNqSarf%Ft%|wD>{v}^rsntvagPcx z$gCjnsqNp&`akcPa{W*4{HD_Pu@oGIb^Ey`={-Ga&n0EP1kwM|d|&3B24gceu@;yT zKqo;A|5?`Z7LOqNEIoGn9Ce^3!qu)yxl&mRal7_mpd#$n)WNsP_BHIi;YfcjWx! z{Pv|%*Ha}}oCNoSR4J7?VFvPsK0m&*^Yet-FRDMY#)BFM=B>hqrIDV+!prHXgy;dJ z*iYwog6r^ptOchk(0^1Aq>}`w?@NA1r)}+ty_~9DYCPuxv$MO%T0yZj3UhvZ9g?hl zCSpp2B(j@?5Pv%f86)lEG6ybQ78yGnkIE~+mMFZGZRwqG?b)CVM1LDqmN^(0$%t;c z!W!s9JaxK6L0U~HF0cJ&Nil9<$3RnLxw3BE8fk{wt$|{?rxi+qS?`zbl&9ehb{F!# zysEyG`ior|*7!r#@HTzas- zpI}HC<8OTP1M<38zYU7YjujeN-Q$ris7ZpCywm{llpK!!)+uePhu~Iul@3xS-(@+U z2D`NfSQEYQfcC{w2N$_#ZuT`U6>Q)Wr5WYgRLaw_&`@*X!L=*qyRx%U+ zRs*y0xg6N2Yo|~%5nKt!h1M|r!F4fJ6%R;sJnJh5A@HlwYrJ;*0#eId*g=e=9Ms~e zn$R3elujOQ^?B`pNOp0aa$WquJj|pAU>h6Qg6&Gb88m2d@ApXze0f3_iw^~XwyzxZ z57w?4`Wb^eflm;$NlVK=Iq(v+BuA1~$AqLhQ4mK+pnMK#s7&519*m*~F6ua34 zqu%P+V&xIXQOznSXFxlk_chU2o9iVzu74D;DZ7!mT-%s_moZV2d!H_cnu6@Ub?}rb0U9w#2$M zb34@mn}x)6vnS_S+7~7+=Pq%mvZ9G%$yUVrTs4dNjTpzFc9xP-4kRJpx&c~b3o|fpSyG}twG%0Wr|-JhQ%YNTqdYks6UtEh|SZ)Cbw5qDv zi*iJugxbL~(hXL*_0)nY`&IP$McuLZ#NiBFt(Y67xN=PX+;Rn0=XhE86LML^+--ov zDl&xO&$j{A@!fq%c8o@AA&vVpuaJF@T_hr#?50+;SZY}2yX@60_Zi#KFLL(OSREGP z%@Bj1r^gMa-}#;cw_u3dqep%^BzVdz4Ip@cbMy~`CH^Q4Eb)l5C9ddm|55+DWH^sr z)nh{VTEa;6NBAV;gQ`L^<+QGwV0Cx7bX2Hjrh=S$;Ynb{EtwfPZl=&czMKz&2IOfh zqbi*gO&Cr}1_5&adtxdmhFU|1d}4ff|AG5XPOuP3|Isn5$-BR>woM{*QB6P#~3LMcBe-ME$$@Z-77)y6NkHl!^WRLZU^+ z(L7I%Jm`;6HU4D(^imZx2tZ?(vG8z zPzi7C`{7AG#(ck!(Q8Yas#5^e(7`QkISvSZTjig(UE7&0Fh-{b-+3lpwqM2yGrf!p z7v}ZZyf7DEgwy3dMw#kcr?QcvTY|$KEE;Z4_jzX7q!)msO^{LBF^Y>{#>~Z$6sFOI zdyOEW*a_?N_$hoWl2+{49!qg*D*hgFqMt2vok*NG_{k4nTs)U0=Qj^kfk-VFy}6 zn~sl_3avf9t#6Nyv860U0HI6j`D6)MnX9&bd6&cdA_&w7Uw%wMG|9!ZQjCa^5-OI+ zW&G`RsUP|(hr6D0?=(}|q|x1X)mZa_=eGOwq2WDrERQdGWWL#>u-a!=p0z6bXoD5mlZ zVJwo%%$N)3cl`a~$yc}gti{>R--zxoS7XR08}Pb%fDD+!wh9}@v!X7^cB9C0o%f-7 zL8cL5+ez~FDq1B!1m(x}Y9pR(v_(&(IFU|y^s@~l&=aB{Y=Id?Y54AO_MSs0#fKXd zeIQfDK}M)6RLI;*w`16_bXR&46_v>orhXdf=QkG}=-OVP`GKhx7>ex!9s?{zD1`NN zx_$=4-WD~9NhaHcWGV?- z;6(24MyVhyzFe+bQnXXeH2qwDeo;B~LMMl+NGBIw)p4V4FQSRq()|EqpRVeb&X8&l z6O|m(6Q%NXB*8soPGx|GBfiO^0Xul>hingS0U*&#>Z``4W9cYI6`6TvxUlrBTzvrf z_;+n4w=0y|+#ZxL6iGPKjnaAApDH$K$1R3a=wSHF$avBh!F6W#zBLVf}f?~|iJa?XD4_alDmL1L#|#aFsu$I$eZbuaKZ z6_jPx%ZIW(!(HeJka4dy-Y!LjuxBET#WugqD{8=$A|1wYfE51v*N(M$> zKMmVO6^n~DNM(HXC;N1?WxhXmB)v^KpOICxmL;%on^-s(jeMvWK&IO($H&K?c+B*2 z|MB%3rC$bpcx8^#klHDz5~0x$9o?hDchBYRB&<0frLA)EldBE$L)mDcVrGOt*ZB{S z-}#-t(jyzJ3**KOMJSnZ)nY(Y0+`-)8n~3Qw8}*{->JP>{aw5 zxAZR~lNJ(b1k#6AA?Q-ZS;JJH{*qZ-czu}s1mrXqHq>Eju;kaYG`Am6yll?I3k_w% zo~uBqCi^9|{;WP2@U|o;?q&rUx*t;b>uYTW_|;K%yTHaQ+y{TIWVVvptuC~fM&R3W zD>D!XA4rQ!t`))VcTT+~_yZDG6D8YwF^)3tP$sia?mWoBaaZUjoBQg;X?a&a8_52k zX0_K%x_ALw#+2q7L6iPCRdMhY!e(uY-MaZRt+G<8A^P-V3uTzaFl*34*rM2+>tley zol6pEh}@7Imz-05^%VI4j;{RC7}K|s9P#2ne{t06QN<#J-+Qa(BQQkt2PK!7A-U%4 zm2y&SdFu0?!n5ca9KNpM$*sA+!|tkjNL+YX+iS?hReFhtr^e9JEDbuBLmoDGdWn~; zm8Gng7XH$Epk?>?en!T@fvK$O$x(Tn$N~SsebK@>j?s0GT_>}+UX2P`ubj&!GMd{` z>Q%7}W#30Q;_vSk`eT?Gpt4H~qQ8AAdQ7BKnm>NI032veWYj;pS9nKWbh3qAFXdHc zbH$X)BfWcdCR~;43?l;3w@{}6cS+nH4Sptlt0bzvfNY3}2>imx(_GsxVx^TD#uk~L zewon7HdLV`CmbrEc3a48xFGm}h5GOO}ITqml&1jy)m^s|{g` zm|D{+M`JFrAHp?rRd7Ry2LhH6n&D~pqZ?=MoKFXFW%d33dDwS{N5esg;!#q)`G?siyA7-~cSSfS5ff5XiA;mOK^za>*x$pevtJ|uf zI{A_bF>NYAI9ccN+FUCkFl_dhQ%M-HQ8wDPNI}eXFgBe&@yn|6z>hRp5yk}9CSH|e zYee_<`kxvo{zefxAsk~&ic{#+9^H2~siQWm{gsifl40i)O$keK zO@9IIZWS z45rgy+oS65opKw7SD00Dw#RDjJ5DoyUrT^AKu8QcG7w9UR9+r#QLMJXd0(WJ&x9Gn3OxO^& z_hZa5*CCTNeESOVONwvil%o5ngMGur(X|3e$lt-$A8_tiqw9+5)7WY^xwB0X1%uUG z=t=`X86;dCxy^2+b1KPO7mP}&?}nUg(@ zxb@bokb1tF<47*Qod?~39FFW(A4s5pgxzc-^OxMNzI934HpnUmsAW%;zVp0FJ@;@5 zu4tm}09DgTV79UpoxK|yKc59_kdDs*FgW#MkIR^4!L##i`GL9!hb@@@c|`NboNJ%E zp*kjDUF6GtS)|zBa0SFw7^@gb4=BBC-Q*Q?-c)e;n+;(LM18h^%IUD}sCAmfyD)mP z`SuMH2u-5T#=f1@;aGG#qt!4GWsWjz4h*uv z>LS0A8MhuXliWRGqbC#iaMspH6V!}Cj#wfWK=re>aQOTI8F0@O|} z1rH{=@XbeX7anI5w&O+=GQQK$lTq~X!Uc|%++HBACZnI?qEyZ3BTREu9{mbdVM_Sj z64m9(i#}3Hy4$YzDE>lBh*d(|oPm#NvlD6ndo`pCk#Uy5n$W%#&T5g3biqjLJ}T8f zJJW5p*Tgg>UIjsyaS`1!@p@Trcr4uki`!LwY8u?QS)5w(y6xEQAO)Eoa<%f07{_>9 z$?m3}WZ{C9D7*2YTnRPS20)!=s*Gf}#NVc{Un*p}PJ?}^z*vCfqxpWMw$&kXf|M^% zvZZV#sORMyry56&eEl+;n0g8Mmz$*9baXkX*1?OFL=|O5E1-C|JlDCFG!{gI>ye{X zR)Ep!Q${<4`T1Thhy{sY&P_EQ{RF#-*5Li&0_j^TzXdY7h{XENs~yCN8R8;KFF6V0 z`#wu5quFsg)Z=<5PyH{6@QU!A*fjG>d80WX!h6Jmz^7rI%~*jL(m9rc6I}4={CxAa zk6rxYr*MCaeejQJd_sbYbEXtDzx>Hqt<}E6SYM2SZONm2NyEGBr`-9I>0hKnx9>uw z^iQPJaX!|eR_y}*N`X%gtWhkN&ofmsi~x!3m`s%U1!b?PIqdmux~RhUeBEpje+Bju z67RkAdVJ#E;6}6f%`?#~D5oJ0z!sxa?^=zZm7D9kc9UKjo!vJ3FIFB3&OZ52!ai=M zJBjSeTXV0J&RF}l#?#L;>K$hOyHf54S+j5lmuvjaD#xf`4usT&|6ns&mM3)GiXk-U zfAQ=^ngyS?*TDyW%+(p_hPg&qOEo?s@yuc+C}l@Kr&@Cr>@i|3pRW&+;HKt`j(xQs zo7FB}X%F1(v5*#SkQmFf)#O@YGPsF{*vLtR>C;QYh1hcTG%5>ut5lGIA@)~|POq?1 zC6_5DUd=J3gjxEDOy^M@+1m*QX7*gNk%97q#~^z`EJm6BEQrf0>XAbsXAgw@AlBUG zn?Y}(X5O%S*PXiRB!G`-kS|ROW}tuZ?H${@ui?I=iZfz!hViPpN|J);01ow^bWIV$ z9F)A6srg6DbTh@txZC*$c5F?Y(Tl0rM7|t-U9fxQW-xfJC;EXJgx`dsgzKA^G+4rJ zO9-XMvYzAV643!6m~K6!Kn2WXpD%Qae`|NjPVoYLT@!OobJf!dHOgrWGK>3EJLY9pu?6F(=VOZn&bu$%)%$1`_ufh5 zi-%>@9UKXI?{_1|`!HfEpqoMV6leR;TvM^q1Qz9IMcV%Is(TJ*#Trl=MA~Uh)psI+BHZ#2%uaps+@VGKbs4<+ z$rQ~dRNsP<6sFD?KfE1`t1Su9S-7L9``6|-`VJOpZDz}SjJ)a(|4igsAnb|b_`Ru{{%5e)~ zUlVZ^R2GN_da>$OB|x7ya8BhZJBmC@$RtV0uA4lRj&77mLpu{V$9H-+eDcM` zXDe=zWYtE8(9UcGglHq!k|_`?ugM(FG7BNH;pF}SIr}7+MjW_?wdtNo)OQMQ1eH<* zmCjc2Ul=sbZ4rycFVsd^!wO1*BB=#z9lsw~ zkpw}s&;gU!4Rig!LYXwY3X^RVp=f*B+yLpY7EWR%3Kym&q2JNSSmgEZ%{oDK0TdCI z%a>D6L!+ltBekhJB&EYNGwNVmLv>AAv}{Yms-Rxgd`?+1p=9PsLT;056q0u|m-lkT z6aLsq*Mf%?rNvZ~%vfEdj%gIv6&5q@!21-R1*Q!=C`XhZa8i^oJ#$Uh%cqe?jkfS^ zfKJ>()w4YjY?UTqS5=0^7}nk z3)Q8Ewl$oGiyAJZMhjjV*v4E`EZ`JoJ_>D)*)#8ZEdHUlEIHzjE|jHZl0~%-3);wnEgxc*hck7^$t{#sIJM(z2H z_rriV2)Jjc8=I&Px#r$bdH-!mV|esYZbkqPlO$;7BX_B1z|-K8?u+{K2nOH7w zn2y`gxA72g{!VprwEL!0RI98ammr;kx{Zkkhi(S$FOABme3d=IC^_IdUG`Vx^LTR( zK8bX0Vr|-_RE%m9V6?qRokYPcP*V> zMMR!CH9ZHa`YhW38F+61B_#zKJ4fwjrjp!~A}VMW^||ZXJB5NcoY}+~4iiU4Vd~*K z#G*z3N6X+m)U<`{qc2@-;(h3{Y#9N@lgMP0(yNv1@oB~h$Y6B|m@mFVSggh=|CC2@ zpnhJv;hDT+GL~=|o6dAZy4w_T!`!`(J@&L$fm}bk|0@tQY`k?@+$-2XQM!;l3_s%d;DZ>GD?GxYz6*~TC<4nQL35O$#zDE!{-~RT*-P`gK;zwqI@al zUtfhZ(5%35u(@pjNx1xuyDL>7;t6};FOSEE;0oS%z|LaZiW0I5bo@qQvcwJCzw#)V z>*?NzP*mh|M+_pJGQK$0p7~P)HGKAHkFR;2OZG}0DdN{SC=)q8zn-lnFM5^EPMh>2 zhwd&AIhpZ*5caKdZByNTzEg&=Inu$y`x!mt1i{m?@N^531$ADH@ za(@j(D_4Z87X8(>c(NZZ2n#IiEpsIQ%S}d4G&Uj;IA9Sz`{q4p7&E~L)(p|tW<8SH ziI*^z$>L!cXwyT~MwH z@C1jR_REo7RQj&aCjl(0tM~NBY}P)Fgfc%ie!*>At^P3*YSbe2!ZKg9dy>vk_qItt zW)OI?z!Bng0F7vhO$yoS7vA)2B=K8l5384l<)|3jT|Lu>IdA5*bbBVDKrr+nFT|hA zO3!Zr=(D@e%%@_wF1@?=OU5lI2@r5q)$or`ob>5MM}1?y9O4(m6tkmzvA-KN8Rb2RNY$-yUNO~?+&EIwUz$BDXw9*oqQx&} zTI8_gD91LHB=&~Var6lNb|SxVXPrY{V$?CTyY5HsD=_z=^+(>#(eP6Gj^1IaXy2f< zI7J&KIkM{PbxXf1JbmRJqTlhJ@aHKnxN)qk=`ULLBkYKK{S~|m^78C%H|d3F$DYs^ z^kxf17*EK$+p=-evNml}k;vlinC=J|@^|FAC%QbbZVBF{)9B^#!2iCFuq^yuZgs02 z1CvLV$9ySfYE=Ao-SI*I_Az>|nW0`Wi=zEG(VHfQGy;Fy)esOLkA1>!ZH4@TKyPX?4+ za;LlwJQ8H*#rjqy+bR}aGxr#r>W((8fxk@O9J}QveV}ZZsJ^%$gE!X5SN*BnMI=@5 zT-Ak@t2_8J7$wbw@Xg3{-(0oq1B)aaWSVJ^8Uo-{|Da(MQ59Mt=k zd%~wjcBozXgQ;r*uz^rkA60FmaxW3xqH|Pl6g+_Y4saMKatOacIpGvlTu*@M&zUo(^;eC>1^ca(O$rZ@8fS8? z5Hx)ZHPxlZwMCI#D74QONRLgxD{Z>A@f&JSuzK^faA+o5&PKB0N0e?#M)R4+T+Or; zT>rq2nU%BS`=-C=5eao6?k38nk)_S6N5~V6e(K`7&NKcc1h*y$)%R$b7<}cz1-F`v z1kx(qg3&iF4m6{)ky9=BFacMm3p|Zxky0Y!y(Sd=H(E5r;=(vLz%jIvtba|T80C8S zxzF=MIWt;%_HW0lcyYVlVmKy%Yg{Jzy-u#frX)qAf^pd}g09uPH*6Vpz?AUmm-5@o z%F5CNnKmoh>>R}&`?i>^Q79|E{rvsS`IMghp9!EdvH_6PHdo5 zd*j#Ouc&#N^07L*rNbNB9yki^T#DeiT%5B+qP>3f>uCJ7DVm8B!xmnK$rgLRDEx$mv?S>qr`e0^q8 zOUX2K;nl{F%nrP6C))(H8s-}|+gaumuF+Lba@oQ8$(eB}DhxRlo?ayt_eR-BvLtrL z7YiC`b_`S5yXS$ICxltLMoc?@JLXG;B{BYJqpK$OP)4hF9RE#r1lZR{lg_c8f{~Mp#nwBT;={YAzfXKWD4&8HnQNVY=X% zA+S-mvvrkk$8qfr(GjkTLLW|cpjdFb(>Ah&qZ&r=V`Rei0Vd?PO+}H16bN7RuwNg< zY@45??f^yH%Z=hk_-iQqhHs;dmzhkrgh&4N-?asfWEsW|msD<{RFae+BxyLc!O z)WUOAc3+<|#>LN@WLsKDr^ICHI-AC27#5`kiIskmp;Hw*W$7Lqx!7g<_F zV*Pl9?hml@K7;67Jw?qrcK@mY(LgmI-1*vvcq<=&;?WPG&Boz9tbF75>>AG0?&=6% zxd=g9O-Uf}Lg5l0Ubp@d)5Y|I8Ygk1;b?19LO+VsPapW>W~<(!rROtq%}j|D5(Q$* z*BIuwy9B$XlX`2w5W(3P!6SsfiT;Czm!7MSM}c?~Bpu;~kCi*ygVRHYI2NgAp3tKN zX7z&MJet`|O5HH&A=h$W^Tc-no=nusK_X+%a$b#qX-*7@zxG>sq_v$MO=z+!A`}jA zaRjslO(>G@9&!^eVd6q;jETe+;}ZUK7jzbC{p=Vl6jDW(9dr|bY@~SH^J!`fuHQ@r z@u;3J2aNPXRL~U|;@oG7-}rUUMof5H4rZ(+Va+-pj9C}HxH^uky2|v(m61sTv`GBm zO;6F)zmMo7+bu-#646_03CDGg5)I?^UKw9(%s)e(HCaHwb}8LA)+)M~&wb8U6*=We zgyvd2@>;#q;j-F?&R0E+>iV_258C&)B)~Kl94?-CY(??7zT3pdt9@_*6Qu5GUwXWB z105T)&OM#T5IN@p3P7W-`_}idV>|oSR*}|D9SBe9}N39%9QqX zE;~b{Y6HmgD@^Y3{E+^X_oaX`#(K&1160!Od|em!-zOk`-cR@h?m z2t}g0mUq;GcvulbvthY5Lh8HK_SGcR3CQh*6J+l?w>P);>Lm#TpZ8_D z`{k=!$42=5>c7(Zw1#VDSuG)_&0JdFpMoY7gV{_T`{|}ZwCq)ggd+?{ID`3i876}Q zzfOTb6>Z~Q$|5XWx%b(~AZLsLMM?)092or1C{&EYn2f4S>~#X-u< zOTW2QmEfd}I2MZ8yaHP)zcQQU`l9fbS0e9L?EkA@W5V3kdw5-)K7kdLaB$8_**fo` zx6*#g68n-aIFB|T$!Go!=zf-S`IcuQ&mh|~u+xSX^#(~EQMQ?tzgtg$s1Lw{w^m`egme@8)=f%TfPl`$YC}{nx zg%^F?USWK>ch#lPN_rr1bwzT2duv+$X^tLM9~?^$)o^n-qNe7K$0Vvhy?>nXbzO1~ zT-`&F=fv-lT1C_grC3+!AGPj&)#xACG$WI{KSxRaZnAaVmGNqk|?ioM=5i zBTcKFob#=_0Gci0b>pAXkHz5IIyW`imq^V4>!#8doU<^fd3G(gXtHPBLCWNH`km8v zAL(5s5om?{*;xzH+Z(Cm-Py+P$O$?&nXJlG`?U&%i#cvGK3Gssk;phBE>S5x=aV&;bS zsWweBK!eS+(|4i#A;uMYIN{J29?cT>_z4^GxoFs(>0 zzsEWHBKTUCN48OL=bT4?bJWTkEDDp$1^)tU2n>7sXHI1k>bEJIe#Kg~YIytR!4!dT zp8&GlDgpitk*SB4g##w#Ur|s}0q|X4{`%#m3xYdGfTcNHQjG~WE(|S-SG>{gJ z4_-6G%Jp}DPQ;`MUKE$nr@{Yv1@!FU?hlq&^qUbxS6bJV**zV3u1YfTn@NmRLYM`L zdwfH@PDIM_Nv!kK)NySUUzYwl#vb{ALUfN8EmbmmXQ@Rk+a?+~FVhEI51}|c1MaZ0JXrKO^;)i!VNp2s zF>JaOt;!<7Dt>+o>JX6ko}5~bo=R^yeE`%pP#d*uUcxavDgwL10sWbzI=%7wfx~{B1zWQo?~1@1^6+USlonj}aAz zKv$qZQ&^Yt*uDCX{$#HMSaU*BWSM3p9ZQ;X`lA+rfV6gW!>h2XwR1%ZBhFPLp?25o z@tskcfzNOK)x)si{WGN{3m1$CAFRl|&2XOQHcwU`|M;2Fy3x{T)Z%^(bBVDxmo9wa zLQrd!DVMpBW=s9+OG#_dOYQ<{ppkfW*VxMNiNqr6!EHS^_dcmTY(2?QJ7dCw;ileT zPeCHu4uNFXL)be9+(l|PzFniPb{9F&W``_)G}C`DwyIyZw*IR(Sh&ie8svivb%p}S z;%gtu(iU!R5x4eIHuUHbS}t4Oh`Kkyl#$ffdM@mviD;tvb6~#BHX1Gie92#cC20ZE+dS68bp!c}VPQOR~mS zcmGC)yi8EJEwu6lpmVJh=qw;{7gYa=0I)b3;6%LTo5Q6_$x zgyMFxPgsX*Jg53WH0#ZuMgA-^8CsrK+w3>Q=Avg1JnG|Dmn;CBGT_kge_fN zCM{g5uQOQ#!wx$t$+lD7+~Et(sdiqpQI3_2qf6&+qq z9TlF^d*@g=M)m9@bElP+m5j_$qL`iDpFPRCuXD@lfUbz7(xmlY9Gzz@;_zxZ@A0Le zqe)!x)rcMqMG}1eyd(&EpQ%Cnt7yv5^TAhFn?#_$;PH(rlpuB+aB~#jX|)kaxuz`N z%VXIAh85GSW8!>UIGwY=;5PBzf*gmeX}V5BdOs_jOva`)PP6$c`H!z>Ec~Q{3`9rM z+gXPflq;R-Zbo<}B8drPA&|~R=J_0(u+wvt>3?t0n@`ocB30>oW1g^o$CjLjm6qAx zI3af>QSJIs3vMt$w(ghh!oy_z9~j8^8QTK1HiJ%8jgi2~3#=J#Q<;ii_d^m*ZkiAO zxo;qEK%$BzW+Gm_I54aM7h3%`W2YN=E^z7$t0P^|z)A5`Ay-KIF?~c+CU$#p%e;rH-4^izf145SIT| zya>ec1Di-|;9%4XnPEE5N3)?SWK6r<%T$m;BgUE`y^3lc@R3fVP%hN!=&T0P-uA_; zM7UW|WSW8;g_`&vi;SDNPr|9I{*1~Vk z|8;G>l|L6Gw&M%f*hRu7ZLVnjO&hbk*xAEy>C_N3Y;0ekyKB5$Cg)fF@%dxs(EvQ` z%VBaoJ&7}n+FiB5O?%u$%#A+6!Xnj9iRa^M5tLnsbZhSH9sc*j??&ysW#*Peuw7Sl z3;JHZa9>k)f!)jKnc!Gr308oIX9#efAA9M#Q#+Z8j3yVMzgEr$nE!Z+xMgS!{ILjQFHpsM0F_) z?6DClLX0AuF^IY!7X*3vMa1b*A|~pJnbxBoQsN*DAMtA0`j)qF&wA~v4NS1KFy;bk zw2YUWH9CGI9k_lTGO#{OXzK8ZzsB5Pdub#LAJ}(I6fHGNA6lYr_;I^*miw?!hK%1B zS47_Q3)CJg&T0?8E{8wD^F7!5!sO)HQcc+%FUw4TV8?&B9Y}|Ize5|bace-C=)%B( zdUIB-$?%G@kWJ2Cnp;?tPZ9I03Sz7ACIyr^aq-YH{E3k|^>9BGf-8Asg_F&UFsTYSqNamGf0uwlCq z6@T37|D`~~U|LmU)`FcDXS%zX|5S|Me&D#H_#krG!u7?D$pbq$AKg3St4k?F{Btn< z1TkR)7S$}{wl3rdF9VqhM>pha*1el*)yRudX9kbIma+REne7|Z;j;F;gaxtkn3Dsq z{cfYDz#Qy6(Rzp9;b#wiV;7`d{r{dVjvT|y3))iRbyVL%zN?M%I!o$x8*h9y?!fg9PADa6J7&-9d^@8jQ#3k;=3x9O` zZEDsNzNCnjWavtQ$2;ONSK|C)dBmT-ud&>3yPtQnKrmL}tU z(w{iqj#)NX0VN=H+kK;e}16)5QTzODnR=mz-hplIFt@frdV6RNO_V3u@WFE z!fxYvDvD1R)UJQJSMguqw|h{8koC=c=AR#6zJ<5J$w@O!mI{X^lV(XMF$$6^<1dgi zE^+A-&4Gbl|L_V*iiSdwO=}2#auyM^emXI7g67!|Bl`zi9{O1!Gw70=eSDZnBK7P^xS5&RoEOzW2cSDT9$x=v z@I_tUMi*d`hA^_-5MX@`^tS%odlq92-p38xhB%T;xed0nSa7v{6=*WmoFQu{K*ZSX zR&_{y7|+T(tOv*z3rPIC+aDx-a~up-Qj4^ko0S7i3${*3q`FoyZWGo=6xva&UxowN zjGONBaQ0EAW`G-fbNxw`&>>JU#U6tk6wd~n!4{Bvt^yEZckd(CAD2JVwwi|Mj0_L- zvbK)Sf-4~PDgy!&_~G}4llb?*pEN4IS{!SQ!z1DstIkjm9Arn^8K>Bddcbf*?GDs+ z{j76pxg^@^0^7JV)d0;ZuK>WwKjL4Hp@)wMKetg0>g88!K-u981L@uhXQTP^p+Ndi zq4pw+Ld*Zo*l9}#ZuC)zRnv96Cd=~#xGi~L+XXVme1}a`Uetd&rh)5V)V}+A;(Ji_ zDwY1gF9=lE8@LUC-q_cW0ux!1`+C?8M>--TD7~OKT>OR;&im{AhhHjy(H|Rsv1lG5 z$T$Ob&%^a!Rv>ttECchGo6@E|vLdtkW;~<^V!wkCS#P%m0jY8>;`+&wMXeC8((y39 z%pOQV8B!+y>=CnAMM1Mm4I)h}ChZ)C-0!ezIGqu0HaC8QN9W`WMd}dh6q?iVZ_-Ub zw(>PB4?z8}cQoWR0NPKf)QjR$r3xTn)_BFHJSM~s${CC-gg`{djsu9O^2I+}>7cK2 zuV3xeUUrUg?7GF8QijvX{-Oo=Hf2nWj>-$9-s6~es5GpeSnu52i2)*|8xdLq(doYf zcv_JwPgATD82-GL`?IfjZkYV>PS2v$2I&6Gc>%+|cSQ9Us9N?`Q2O7Xonhq^6c|o2 z@Eda9jDqzlD_bI$EL-w{hEqRCT~+zBIU1*3k2Udp zDItg|cGd@NRSw^3qJvVGxOjb+)JM)A&WIsuKL88I*%{2QXeX|px%4u4_nYDSwnA!_ z>sxQW*CO19bJ*guSiFDxv-f~XTPy?-tqV;;9YT0{3+UU;v3b&QB4&QYMxN^<69iFf z6>UHVD;L5bi^%LJq2(ZxwVLT;8^>oY94^D!%71QV(|0#DD)F>%F;l_-o@y~H=&w7H zvoT;7n!SJyxnDaE4Uv{u``Cixk=_hxGcfSpPE7|>)tT?;k4|&7PyZXC^9cs?=|j(9 z(a7{J*A9uPXsLa?R=VvJv>4YQ+Ol~XlQNaqeAF7mL<(^DSD%xb2qK`4$ZY>=<})s- z+wIx~lEe>+b3L1%N<8217<>k>s`jQAcCgB!wEBeCZdpcn!~-;*bIE(j*W8co?_gqmxz+b7YxJf1->(H>mFp0E zMiU%=0$*dNRWthuNMKP?fbq0pbpps!w3nRHF5RjJw2qCC*#71IGa?nRiNw#`yBpUb z`4c(HZ@3w{RIWGCu>rY+A>n(MbNG0>ohykLB05s~V5t}5abi4T>iGU5cEzPuh*atL z8~*QHo1mB73C_)Hk$HW=QQKu=;%Bnw5om(uU0^N5htyim$r>~rU<<6#uKBxRU zR!dq;d?T+B3)R(_PglA`8}`4fksJNJEt%Rl5B-Nu)7otAc}{_e!wHZ>L{9>1C*4>; z{cm7vk$wySXA+0R1B(UCKky?=Bo9;BVlxZwleqyC=-y|t8dc)oT@UMVsH*%+Ir-I0 zZtMrbJzW77zV+?^Fp~VSCOo}~>k!b-7IX|Qy9*u*6Ray*w!GE*mg%*Pr7$(b++8m{ zqep4{i4z6$C?gTWW1ZZ%K6ie;?&lC#)FQ76rM!UiP@4TE&_5UX? z2Y*(hFltt970`@UaQ_*uu(t7Cj@FDYP!_etDXyYK5w_s1j-k95S*=8p? za07EY%UWW+To>qv=(Hq}HK2wut02@7MH8rM0!!%R4JaE_VzG{61=Byx?^grOTa+xE za0kEhMSb>nZ;zTF2_8Ukcqr|~f?T+n?Oh+nxGn=z3zm`?&Ai@lda`eNd;=5yge zu-UcJAaf{U3v1SJYN_`YJUfZ3-t4rM3Nh~-SPBJ7B7Wk}wMo!)T-%XFEVw>QAnSEh zAA#4Zg(cGsi7n;s3&dedq`~U$n8dJHu#dt&x4UltY~x@Z6LR@Aq1i~SAO8U$zwcod zct&3b|81Yy1g=F)hki6ZO0;fSV*srxGzf6W%=^)Nbjo5MUUOVcmFRjvmMH++91XI0 z!u>mQowWMaciEKwWfuSwO5%)FWjMH+%F$`4(0^~LFvcMTdwC4?T!*LW9@fZ>?C5hP z>G$z)6L2S_xR2etEW3w3`8zk~WL4V6h&1D(`glkEH6xE-C=t7XSMto z46x)U{yA7O_%L>#o%OZA=gF0>%9&p^8xd~jD5F&3{V}>M4_TzbT|}DbcM4TEuD0BA zzBRQ#6x+C#WEQdfL}y%Dh9O{T53qRzc3G^{n4|Ga1{5}`i~ozCW9X;fzpX+f>m5MH zx>fYB?wWhMaC)?6NQO0&wydrJ#c^^qUQ^SDxHqKy@?$>nn-iaj3BPU+D)lwuqh(pU z5Z!#$Vh7ET-=tS+ff=DjrJ1KXhPtyrvEmT5ebxVY>;kmepWV(3smsyRnzVbqDG4=R zhy2fZ&n;OjnJr;^Ub@ijRMW!_>-XwPgh@2-UT^B}NzAd(x(s|4)f0Tik?z81B zptNOuhOHKqYenC-2o_>)Co+6jzk%VeRPoUd{yZNnpJ?L}+splp9(PL3!n%QNu(oYU~#=yymPPBePH_8P{n6iZa9E}ojE_}^3alk||6q<7v zpsl8&(Qk!m0&?ykX`=`P^zD{&@TkC8ko$n|_!>1w7@TfNeWiZYVB?Bta@RaR7S#-iy-iq8QAFFO@^pz*}3z}vY_BVA= zlE;0t+fd}!-HL|I@RvP$%~h&Vm#m&$jy+gvRRsOv?X)C?aEYk6v@ zr$nD}VS{9cb%*V&68@1i*T-NjADw&YPCfXrt!sgCK`oPNhFdXRJ>8Qltu*1rF__rv z4pCNTr@AKyp8E7Y&?$>sS?p(qc;x?Iq`hTal-(Nkt%8I!C?VY`sdP%?08%2-Fo;M< zBb|~%x1@j|hysHm-8F!eNFyZ-jdaVqF81CZp6C6t@BMtdf8N8)6>F_Cj`M$TvE9Q} z+fpt$_37z8GQ=#&O{~cD5SgdRq+cbATC_RM0B0;ORuVCy)$M&QNsnDy_+G|}#rXUX zH{~o@z3iZ@$!?W)&@;%BsDG=_3+pB^OTYFKB=-IiwTym{1`E4cA1Q8TsFc-7w9?Cf zR>T@%iGrv_gCbDD+d|Q;tGHti0H150c#Gph(X$+newJz!xkd;WB9}SBRCPyX+ zgvf++fy|zR&o76=0L=cCSzxM8AVX}*6j5XY$KASsJ`vvv~Li z1Q+&*F9WIau0pqHduXsm>b3a&js8lIHNwtG&12>(VQHlhb#u`4IyF+_?Qj@Kv)Q95n6K1B8<` z+W;?lBE1r7f7w2cW7GqSLVxAQ<{9Rn(i0(KiyaY`$q^E)Zk4dUcY<5t=G@UY5?CAd+-jrb8!UA|gzB+tTs*Y5!U`11FuG%YD%*T_aR5^Y% zd-@S{?y@4*PrBZ#moY?E$6LAavtMWN$=*3QU15AT`Vxx*SV&!24DBZLN&!<-AnvX& zFY7LcghQ>}kN@B1c~8bu{*BO2xrFr|Q(H37DQ^DLQow9*5-w^c_!t`%FIl4VnH1lO z(%qA{1j)4^iNa*nUfy`>u4R))gsm*7xSTWPD%q(jxOh9-Z$mS*$q$*uuNp2yI^e43 z-hKbty;#H+NJ>>UeKdbXE72pE&n6 zR+3)tJlh0$`$M)-SIL2&S0J5g&HgN#oU4H!EY$!H()3b;n{R$T#$x2;wlW}ZtX*0R z2w$8BxhlnMFi$b8suPKypl%Q4d{3O@G;?3I??^RGJJ)uO#6G#{RInJCqvtE#HSeR= zX_vX%y-^Ryx#XIB7g<41YpqTS%erurH^$*A3X}Y51grWK%tGP^i|64|NqX!(nCwcY z$9Xa1x&sUhEXE$m8rdm$0}>5{K_p`$98$r^NcU8>J*K35=-f`@xnaLSTm1bZM(2ko z)ZzjyT&kZ#RuGQR?(g}@dkVS1$G84=IG{59NrT>{=bHjBUp}A8WVUog6EJ_Z5n#yl zxGkp!(P>`vvIKU4oR98X>diCaYa0B-6T49#BXcpx%#?h`1tAlQk&xj5K-j~)9Nf~; z5Mw>9%e5t=fywfExk#%tJiRV#dk&`zn*kWg2@vD`PCLewxTBg^9`AG!M&sOFty7Us zk_&q*Vfo-!7mJ6*!J`{pMi=%py#72fn6303Pt(dN@%Vy{V`qE*;DlHslH#-c+di&d zs`Za`9HA28KE2$y$@g0BoC5Lm>su0JQ2SyV^0wtW`MY@Wm^&W)8zbVg!OzJ9&xLN^ z&KQou!$+=w-=M zZhuKOvx{273JaZ7d^hX7d5$;H%buA!=>=X>jq$F~xn_!We8LZ8huG-A>R4L>vK1#8 zac6!)4N+lxoa>rdE&#fS?>R%@QZz8K+^a)zyy+gz`)yji&=uH3f3UNrkvpY=ns$uq8CfDjpd_*p zdJ5bba>$rJ$R{uPLn@Om~P=v{6y9v7a}#pJphU*I-ZnjBoFRv;X*POr3fa$9Y-Ivxqqqm z6mwOvP}8ddb)Xu`T^Xg=l-IWgP#!{p@BV|)>AoFCM`?hYv7dpAK{zeYy`>&R`e;k7 zz5}BR8j&kmpf_$c+61BSI8RfDr&l82a!k%t7bz+>NEQ3)`B3%Rp-HTU2ShxN40`tF zauAjx@f@3~rq0ThxOEE4IGxD~iiui$@YUh>oF8w4w5)XrU;a<7CXN+lF~01ww3=VS zop>HycrVvlE&MqXuzpTcaoAtwLS@lvlebvlc7o+NDR6fbB-Up-Ij1AKkwTs9R^$zc z4DW`*w&9!IndAn{HOoue(iiVxT(h`H-FSnj7amA$Qh^-mUpiQPf}R8qBMq$3135W) z8`wji?XPMwOmUjOtl+jR8Q8*)vlFMspqSY%QC(PdDcY0#{+*KK+>Afd{aJS_L5l!> z^7~n5L$tN3G-FD31N?O{S;-BvOC|~7vhi;ZJ&Jo648{n1&%QeST(yCZvXSAPf?CZ^ zL=SIs!kOs9;wHBD(cGntQlLtlI zW83|P!XyouoXD94OXzT`P@l1tP|6<^s?DM(mLN)n+UsC6&2NKtzarIX?L+_fMQ{(= z7@J>XNC^{-_h7rpRo(eki1R7e>i*k;n7cBWfEm!J=IMk; z&Ga|++yN6fqBl#-bG=*eOyc%+)VKhpS_Xzzt9c71D7+u$M8@EfJ^AZ)y>m!1$j9i9 zj$|qaZp2U}4^2h_zhh^vN--Qxx%fr{5$Xi%zsV;0eOc?zYOIys@OpJks;xMOaX`W& zL=(q*)#4nSDs4Y8dr-RoBumeXvi7Yr$#DCg z!^oTgPrYixZ)Id@&W@DJoElyqyvJJ0e99FQ);B>m{RQOkN-ms(oCA&-q_@peYe5Su z;X+ur`5m$U?Hiv4S0!+#bPvv;;(u?UYOd}?V}xJ0Muz^LsvH{g+nY>K`83G>B;{Mc zhk1*FwM;s$KX{K@T*ZPq7(-aZ4vsL6)V32Dh!JZ+zpMVV&05ZL)m_l{fa*!oCbazy z_}9(vw=yUcI8(j_K4JNFK0NG|I2ha^-3iAgNdt`(*s7Dwj}h_r@ZS#cY7HvnQxIfS z#BijABV|`*&CRumf@x!tJ;l-srvDf#)o?25DpM=?(D)tC?+7$FJ3!1{J&jJSonUYm zd2Y;y5BJNFVqeev<`zdYs#k6dYC|KKgBRhytuljlxcPn#SL2DUki4UL_?XZ>Zj+zS zG#wELk!=#lZg|mYH7YWCQDV?gc|=UIdb763i2QnO>AqH6R47ZGmqCT6x0>5g^?Km9 zh*mQUu50DY^NKzvf`hH{wnpT}tj5ZiCpt^SoGeomb^MeSlX|{10)q`9s>y{hSGtq6 z?Za+49cH#7vD?jTs2&m%O&G0VXf{Fji*Eq zjqwp=O5es@AJ<-MlAxiD)K0ySdHRR8dBxaWxR5{)W-i&~Ny^^i_Z#B%N0zX6M{k=X z&YH?}xtM|rKVg%7tY{)Ay2>Qk3sR=K`(j;{TVS@KZ;l!$$K4R?4?9Ql4J+_j)@dNd z3-yf=z8^zW^&3S*xwX}?v#PedCq8A9bTMDUHIBXZ9eG-rPRr7vLxd=D;p8J+vzO0 zr9Z)1(WV_8?*u@oJFI-3s#SpfsQZ2&vXofEiV4P4kH?S5wR4>~0xeU6QuZHA=@dg~ zDxo(G0{!pS9-QjCZ?NRz)bUzYl>4U~g<%_&UkW^FBY(sH1#{v3&!qI@NYElGWKjDH%&3D(3+&HV?Mf37tL zv~#om;y54&vfj9-(xzboCzLFlVr`g|@poZ%iBj0|g6Y2D_m#{oh6y`9ad2qB;FkCb zfqCazS}$TR_!&m6n&mtJs=IX?6N7;@K8+Z#bRytaWhH*75f7`af(mPjHx$PGz$B(h z%q;97nU>+5Ztbxn3t>8hm8eKkm+}1Xuy{h{o36ciaqn@8$}Bu>Mun&X|59qadF8W6 z9aX3PI+iYqE;3P3K%y;#;1M1_jUX!434!k``tk|xqMq95oG2DEz9Z#S`29{#jC@9= z9zZjNw~7tZdGVvw__YjLaWRA`H#yiNbIBGJx(b9RMo=0d(lI)mEvhTy0x@iOl4Sec zxm|eblmw?~nb`gDu<@qj4!(X9Kvk}M5wtJluyyd3utzRI#W(ZOasR@KS|}f)1v7&r zT9k~i@7Viszxr3L995f&6j8ptWpEF89u;lMwKo8 z8TrQu1OT-hwORsrgCT_HZgONw`NKkQ`=ds(qIw~P2O>ds1%sy2&|6t%aYQRSzYFMK zL*=hw%Wu3!yi$?8ZAq}NmC0IxCcz0-ui|lw35_3$SL5zW*T=15SzfE?P&(Ogl;(L} z$-)pxI)eCwlf*th1oRJrTv-h847j)JG&Gi<>07c^>_x#QjO9OULl+lCa~y_&F5p zk1(kz(ZRHYS#@e9G%OQa0F&C=2uiDL){oBZ%8T6-eH+%E@+LMTl_33#=}=xpj(k^p ze|fFI+TzbFWgYozO2l-O0ySE8n!tc8dTHBN;lOzrtRad{s|3zr22$!^yB`~3@NzlT zW*p@RM!^qv{6!QpLBG>fRY~-Ad|JR#8Z6YZz&6#5qX#!vR{(cLwVIvyre+NT=FRXI zuK%u!19DRP309aBGB=*!lUnz~P6ii{H>s1o)SIe?fu374oXpIbx=1{8GwryG>Qj$r znX!x$K8p}j&>r)WVNXyz9=i-?zLDr<^0$wt=kNJ^IrDEnDmvqQPawdPH!zyK_xWXx z)w{*k0D-5*hdEG_hI5KJaU{u_Go=WQz?MOTQvP-PvZ0hFX`VoR@vXFO@qSp#ZRMf7 zxi5hgd`QbXT{na2`0ktEU$NAFLY>dINOGe(a_CTz@ZvvtMp}SnUkGCs%!u-pycu!m z+S4aTky`27%uTqBGFD;>;miqSPM%^ps~G7Rl*)9D&DU*UPrqjMLW(=YTUoL?oeE8L zR~*5zxEUf^1~oAkqE2C@!nTq#D;;cESNwXB*=fgs%$< zRyLx9=)3-Qzcn5;e1Y-qazz+3mp||<4=||txOXqhW&K%xzztN0Wrdw?$Z<3&$GK(Q zx6=LQ@8l_ATv|=L-Ih+AwF6aGvR-WJ4FnKv=HY?(`w1#u1i9jP4s_RIQs|Sx``1JIDwsZkZg+l`Td+>OkqC?iV zhrlvcHRdI(pHY%BisZXnm&eIIyCY=mRZyVBp$}o9G(|b{g4@DX*m8kZ#I;p;=_bU- z@dC7wN0B-wB5jAT{=dL$NHr2;wM5>+QzGmg2WN1rKE*1@<{eY5_OuVoUN?iQui9Tv zG2e=+IM1?HDoS9dRnL;FhPnh@{X+?f*@ogtYcFpdkD=#Mvd(h1IQAEJU-rmU*a8lg z&?FaLdU)$QnlKSIuMD*$gZ6OcT#aPE$QC@`R5Mzdw(4yQU0WvD-9*ON@U4;!?IEnh zWy%Hq@tsnJ@J~D|-7D?;@v4_4u!l>&ks6w{oBv?}c2iTEy8(#qYfxA?|}0Z(qPw9WB7RP}n|M15$5B%i^; z)(A51CJy!smVOgspcf=Dxe}_i@niLC7BBgk(sLN4;Hq(IO9j1DZ~zVsIf|zr?&xU} z(Aa)dH{-S)s%7-ie2r?q4pomc^{>3`Mxez zGZyv8UtDO95MobO0scM?~`2lvxQO&PbWzG)ftblKH5O+k>G7izf?!GM_dEh8<;8M&Fs zk#UIHgtj^S;A>R10deIuDzQC_^ER#=|UFU#Fz{8;rj&HQQ_ zLUIU!Qlrp}DwKPe3sXPtN~Jl!oijd?%xRz#r;9trQcs)b(*eZ9an7q<;c3S7o(b{P zJL6%#c8gseKbBH317|GG*1(UAz`1@fX^0^v5N!R+D$m{QgKiVCq>affHJT*GQ`0A;51ass?sbKMEq zj#iCY%B9bo@!d!HTy-xhPz!tYhBE>^ZDe-AbVjTsX>H<}VIPCoYt8T5Nip}o6LmMk zCtV`5Y$&@bE|3N;NPVQf3O4@a{RnrhB{p%3_iw=Q^B1{s35t%q`q0|cELux2AZfg zVx~_HAR?xn-X|7AuVX3Vq)uzQCbyO!Su`wBU2QwG)_#3~8LiJQ|B;j!I=8v%ph?9bId37Vv|4f)VvnrMLsQn8RK~B7XoN4$U17t?0O$kql`- z=y*@eCZ)^Pj!c|QN!AsW=V*7s-7&9#;D+PtO{F{|>pGs5FJTotnF5g-wjGw{qV1{y zsxH~=@m7K~^64w=H|_!&qhib}uOAtAqu09$RfKQlbr*A$1ATI|t^4D}H-!k%9Had(XWc?ljRvkph7sUMN(zJl;bUL_`#+DAV={#C_p z=Qs8X09*2iu+L}@DdfzzSsM$T=!KZoIg6#$pez2qG*ItsK25etPVK?iL}}#N4A7w# z_fhv&zamx9BH!wz$tYzUM>-y|2RliaXDUCp?hJ+`x06X51M6rxrr>fXw~s~fav5jB z4@c!nT9a;~HPx&=N>+xy3(0)O);J)^HJ0Wbix3Qe&v%lKEA&h5_35_O8b1hipdxrk z!Q9I#tmMcjAiX=yjL~OG~v~q8=i%IhPN2ORqS$KA*Ya(7%*-8 zXK)PAJFK}SM5hva&sZK<&;**%-CuTo2C8}%T0Z=e19Z*Iw19M;1S2smnI2y z{~MkmDyM|Mhh8MJCsO|7e*(ynbXBlvF4!a&2KmCje50^=GpG%RIv@E zD)^b#j%2cjYJ^_fPkEmFsDK2a_NkhK>f^lhuiv|UaRG%f^{3l6pe->`m4VMdGKD&M zn0_0Zu||t?=BPSs+2A(&{dCQqqJx@w8+>2E5N|jl>N5~lKb`inngSKMNj9QBn|rY$ zPDYd8v-mBY7WKF0Y3S){%#oXZH7=HuF`ImVb=YH{FI80os+C>#!5aO$(>rFzMv`<`Z-F`u;BfhmO5JG5n4W*IH9tVEP|zfC~SQePy?unHSW;6xJg zPdI4mc@$q{?;NJ`_N6La`{*P^ij{e0Ja%u9{q9TdPd_2-*fU|vPDF(y&gZX^zve25 z{p2@0RVu)F_4ElYK*u(bA+%f1Bx2J2D;72%n<)F%*CtXb6#TwPVHB6;vy0;F3D~&X zv#T7i>z(8$7yi4OAmJX4g6p4)Igw;)Z$kqm%V3`y!gU9uhLw;9r=iRD2bNCmV34$E z#j6K{bSPe8t`HezC#yu0R}H2}r*GHD)h+FYYrW*v7u8N-V6q?`(hxp|c*$*p_jfzE#m#ViqHQ{#Ugs`FRGRqF6vDZ>(JZ(5sb!&dyzsnd8)0WzqXScVa}8Ml z?G{Vl&v~E?eFgJFQX;lu0e%dm&#moRBJ*FZhBN&BtE42okNK-h-1RQuNnm>RtDs0H+X_a*WcBILkM5EL ziLLh=O|QJCrjQT>&MjiE0F&OJZclYEiIXjdGI}es*oXQvSfp; zxHM$bmtKC=%hzmqvu3>oArFv1LHE37akl#J&O3UghBXT~p5oYViCR#}e-6A@GYrM2 zLkLsI_e-qfP)T;#E6MN+1=Dc+4H_m74RvwSfgR_QT|Y{KX9FDbD)iW3A^-e*RTsdI zk}W%oalcr-uVMeKY#kiWtI7chqwvo}j6Xz@?bZrF7cRXI<(ej;Tu3_7H=*pV;OY_+ zFJW3(LE`nm3$FEtHCLOcb!Sysi>VjRlTYI|bV${^ZnAzF-QP1F0gL{UD#2118UJN( zB|o;Xf?n7RsSGv12RI(l1$t!Hl?iIGi^dOBmvdwu9U~5iayP*YfJYM5J zy~@-=RdYc-94z(3Ot&wilTRxO8+A=q}Hw6anYgq&2?`XIoqN+9B zNP7<7?2&7{UU8*N6K!cxgZ@W-TRjY zugIl9a_3RR?Mx108Uf~NwtF9r!}4hp_Z3?>NzgFNqH;DYnNu8WwvS>&G@NrS$Are| zs(#fLX51PUT~7IS_jBfR!v!aK<=fubz^a8rsNpNO4Hl3j%N@?#UXf(h09w z)e~c{zR_yx`h6xJ;nY|JWtD=iO8#D+oBvvkC3wvb22RucxPAPG7>|GbrpgnXRG{SB zG3LHBIaZ%*m#4;=6>3x#P5I+3@2Jq?iE&Yo7$Q$FwNtk!GOod4s6gfRud*}_aY#U- z=%M*m)C#63f~+q}J;Ix9@vGzCd?H2c?pncqlD^)<70hW7Plvryxmw>}UtHn}8cghs zE>>**$c^E8PeT(`*(57Ene1f#$}~<2_KvW!0UuH4X-WzqPt*hXZ*qJOi{{fQuJixF zH3zSR^CWMI){R&+eoVL?^))TH0(S6K>QDu%KK)1E+Z9D*F zEIh3fNE@i_m#6Mm2D_11oPZh^H+Mcl2=%L^tGB)*WhKa`RT58X;%fQRB@p4mNk_C#Jy*@3YLM(onQ{a|z|}ul z&&$a>nYWqbd0TiybrhE~ON1TGx_dQQm9bnOVxpyxb36KUmx|Pw?Q`AARjBb;`yuKf zj0i#A9-uf)KnmSS{puaN$k?Bu7_s3$UFn2r&o#X*P(FVyd*2zSQxr)H`BJ}F_vvZB zg}xh;zY!}y)Ft#;)A$MdX-=w!#;GWwo9JBP2{OV}!X#YS+p7I<1Dz1f;p^<|*9#H$ z21+th4Z>HIDes>*9dFHh$EBvcqOt-DPvvmd|1#-&)7DYzs}TG=${6M}{4RJSisun5nZ8x*8Fs+)KV|(jLxn@Fo*GXeH1jqA>=@`yXsLNz&t(6mB}- zp*+N0MW+vhuq|5bCs9)5PuYR(HR6>q1KxFZK*n!Lv!)bQ&OF~E^&I+yTXrFs*W%z- z0cNf8Ie+Yi^pEf`aAcK8GG44u@moC1im*f$gz{HB!eX#%T0$N8Q4welg#RTsb&KN} z{sTHF{8>2rFSkF4cd;#wlw24A6g^Dy1`CD7M&K`4vxiuCozIg^8iI*K0D6Q4TI%}X zH{{4N+dX7a_X0J10|!Ch&UeHO85P8VReE~|D+!osB(F14zkRkVaxgFQTYRk=aNB^P zEnCqLaUBPL5;%sgz10wxl*ZC6foFD$nenkjN#BWb^lIl!a#Gi#%zykK{D4L#l9&aha9HG3X23Bm&cvCmzG!0JOo#Eu zysH=3nTqD$30nvgsIsxCUl7ph7q=2vEtGKpq#YGAO2D6~))zEyxrkBaxO#v{mE4HG z-JbaaB#2;3CbDXE)*c>bnysRlwQTiRp%0tjPdkmUYo zG(Wc;dtT%=6GI{Xbqsz1r>RtJTB}+;czalaS5DtgM2Vt$^;4{S^dR+OoIdPb@n7#^ z3G6F1%?Bn~CG+e=#HVtCKe4T|_jY#dAyk#K ziSm&jcVL!zcg|*k%;=}qx4>D3{S3CTcP}%*gD(-TH6Hz#-Eu^om57!OOjF-zoH_jT z()$3x#l;I^`xK8G4U-Zn9uC<};Azv`_&1G49X%iRBXR3l1M`4utluNz*Ltj3wWE}a z#JbAbY$aw)2+t4QK`wB;<-QkDh_rI1bt!ivaKE|*X<3xd?{R3}FuzR&o!)Jd6wvYb z-=hIQFDc;nV@#Yk^6;JW4{9gSjK;mZ;hsz}m&lJ$uI|c^8w|&m4K|T_8P=gi|7x(_ z?w^et_i|kqe7g;r+nf>?zz#D}Eu^a^A?80A%kI z*M=EHu4pu|A}UX+016tM4N$i9UtCw9ReJL)|Ffx$L3yI@jxp^gP9THotpatJ4NyYq z?WQ~dZmrL`OUoa+?{rXDS6UC4q-ec%Af>zo;3r{0UmJ&#pFHFA24&PW5z2+PR6@_; zL^9`ph-1xu{FlGr-@jH?1hccV6M3?Swub24&P{FcQA3<_3PcE9_l|(BVh{n2 zknf@db|1lbm=zeBr+Ml&gO2YA@2O! zAT*K1#);zIv-|tbQ=&>+66OO*#G(&Uj4uS#-nU$LFJBr_*qg) zCg^G6j1~SbA5NnIw@8D~BliD=MkD+W;YC+h_#YaLx9&f_VyKn?Me~fUUv2)E53_|q z*oL;5e6R-p*((QWto{xEQU8B||Iq3E^DkxyWF!Rx*Wr%#&;FMWqZk3B;kV<_hkty< zPzmne2%dkqFZ^p4gAoQ=^~*T}A#I}k@c(kp3;6@ng{(E^e-Sj=0`O9H@xA}y!4=Yg zWs)jHG{kvC@J0TAxe5E~X6d~Q6Cxzk|K0Zde=cRSNl4H(Y5Gv}e|YJDohbPAy$V+a zh=2O|<}VOK>q}Ee_J4J%{GUHt_J$SMAyOiap=QHB{d}N}CRnC6{A_aFVQ{#FlY z#qf~rRQ@+-;s5;wS7??We@VUmho5~NHwo(FN=N|MIg*IlxnsLXi~sz7J!QLMz4VF$_>3KZ#xYB; znG%ZpJ0wVSwj=r<&Lv7#@BoxmC1gecFZ2HHM!ZubJl^McK5Q}ns`>cOt~t;u=`IaFcNT}ki-3rifL z(O~4tmHn&nI2JJ`Pb8JJQ!ZOI?c8gZA1%E0C2F0OKd>1*&je#oJf5KSn2w(T4S3a^ zG%=Uw!nUKO8=RV+Jt3PJywwosh4oPi97y0b4*{h?>}qDbu_(i#*5(J>v#ENOWqCJ^`cU3E*e(v%>q^jJu&bAjuYb^B<{gH9=?` zD8h7`9wUIR9Cqgh#7A|9Bq=}uPW3bu$FB~Up$WP6fIclxhLpFv{nWHKgcN~85FK{_ zd$IE06Y)>{RM{^+foJk3s33S07-5;sg15*TXl=eVLS^|3SonhS~Uq=M)$LXG-krvAQA zEL-3Lrx}9tw7AuSLfCMPz;hFTJsjS11AV9OPoA2;aVk=564lT#zXW|bjq&=7Z7ol< z93rE(hF1nq5h_9V@!fPZZVvi;#@J7fmk%)rC>$hgK-8>JohxN8@L2-b7~eq#-KdEt%;m`^FRAGp`}cl z4Dn$c#$dLzooY!;G+~aSz@M*i1;p&IvSK)zJJCRZOgv0WgHG zg!X8$y$=W=O@vjWJQ{>j20k{@oBp2HMG!2M997XU~^KtO0}BZ)$6Pa0dba z7*`JdG{;+XAGY_m0i2%A*V0nDr8yhX z(fkfpWc`ZLCij%oJg(*kW0~+f7w3N1PCip$A4|D+<3dtKJYJT)MQj5&A)y?v$-m}J zweRR~$Fg^^wZPswZ0xCAiIXSfbgx_V-Y>~pw2&D%7p1%~v%{Z-F0Vp#QOuYf#*hPs zQw>(4IyP~-W%-z%0#O15lw9@}niy%C37^sj3Ni+p-8TTA4RbhHA1^ze9Fdkt-bbwX z^EJ(mkLg2acQ6Zo&)E*cnY|%cXfJRwp>be9fEjB%{JH^hGnUS~6iU^{_^kp(;B(&IO{6pa1BLr^scNV^6P+Og5}e%AJ+a-+1eU(NsA z#AH8@;(ebKhbhMkfp>L4SkD@4S)?Su08r!=oOJ|KYYY<+Y3q=ODXG~)sNEgdLY&wq z4xn8b0JVkt-B$El$?)w36V)f>+kQuNsFRyT*#J0KI$nC!HHUEpg5%}>8F+=#GMUl? z_+px{<{2XLscO-Hf%ef=vwd~jV8H8%v)!VJ0r~zY6} zi}hA!2;np!ZuLL!f^)F&zknJpXA2t<7!n!;nEO5(mh_J%!pDh<2BF#4w;y~voPu`l zYNTBL-M;+gEVz!Y*#Rav^bryfEY>K%<&>YyQ zx`LS9(4~`jX%EAMrauH)cs_eN`fb<^o+-Did@RtuqB)oK&;^tpqjbK#EtE(~& z=CKpynPXxFQpUl-iWC6=i;kIL@27JGy+@>BnvFtm%#H|RC0NX&>?z^`w_XTm35)b-{E@#%pHgr-PHY#-#2UgHWhSL5r zSL1S+fm*Y%*hALPT2y*;Eens9ydgWF;0+!{)#kO@S8#U-36V+b(kEbZ25q4z_csk? zE02b&Dmcrvu~D3kBpX4PwSeabPi;P(@%i`kP0r=T?B(}gJ24&w5i)E%ug%sN43vA9 z+%N$VX8ZkyWYW;{g;Z1mQ}-9 zo1ygg!1VG#vb5b8L#KXT$D%hd{+4(K2?CVV1Msbd_J$+Ew%?M z=DJQ`m??=9kjJvpYx;LLrvFvLVe0OvC_MwEH6$DRR@*CzA&p6{Td^d`|7T&YyHT+S zE7L?%`s8=grzoChOQnx}L{4IMjo~~xeSS~-yh2)%cXQR>*d;+?tAyYsZ0SFNDLd&G z-WPNSLJn&Owe2U)gy+AyDMRNeyt z6IaV#!)hT>%#iZJ?{BtNqx%DQ!N=XZa}4 zR6pyL$l*kBp7s9T*(Zy+tGSyC;xxpWlgx!1G*q;UQx8U|)spEeW=Swdc>_s)>JNJ3 z7B+(XM0`?c0m=OB>!s`pnyD6mqkL17=R@ZT&M7kaji~kTjnlc;u64+c{~_ zt(WHmmwmjoD|wBN?1XUDm2L{I2|5|mc_(Y$E&2)^5SZ@y4S4@bghcL3>e}7{H5U<@ zJ{Cq*t!ed&13okBCL-puX}>#hW4+yVS`0{n>8MCG!2qlsfcZIWPOiIP6nSQ%Q6Iwd zEt5kqenY|E4iDEYgB1R^7W|6(vp^FP_0e|rXmXarFxrx(AKQvC?HV@bXMy+)c;rJ4 zPm~K0yFq{0-n~Z%^Bw~hDl(9p4B2GL@iVf9_5>wz_7^Vk#4T<1gera`V*@9ydaaavfh5jMM5d@q;LC46K}Ro{r0G|#2w(s z^=n{Tq<`jLQ2-M^)zPBd!Hhe|aexnKD1}@DW_RdjWeEiolHLkXL`N&o=}oDpT7mpI zmD@^&_0J{9OHVEh5WkH*hwVHa;}9E+@Dhu>7?eE7EqC1v3kgXLluy3L883N)rxvo; z=1CgNtG%=#Y98hzCwjqu;x%bgq^Du0Ll|o($>ElST}pcTMj%nuS#X+|QND_jmgVCM zsJjXH9n&9sYE@uZpht~RWe_Zwf2zd_?*zcKj|suAlNU99N<+;Q7T~n1chn#!s*<1{ zc*bZ+!fRX%r+VlP zi|KWt3LCaqMm5>Rvxec&FxNVHvA(#D#_QxW;-+~VHN1ZGA;(FlY>f3ZS_e0VrkqXw ziAWGe{|P!LN6<|gfQ(S@ZnBd#ZqI!AxxM`@L%Qx0CQJDS*$Vn(C`RMwMO8%bZwB!G zY(nqP!sG5eu>JcyAN~<(IcJh{;z%SFVBHJWcLq|Ys*bLLP+472p2_Nhw<@6>H4^Yy^1T*EUv880k3K}i4PKL3WI&~6cv z=AAY1aS+{f^#1AD<}&3~OD`8U5LHVrvYIgkB2AsdW-uF>aSyZ_rXZXtu%Xj)PjGsu zYuBQo#Po@{WQAWD@PF2eY7U;#l_B&&1X{*g$9}{Sl=Q+D9P?Rbz3qfRAMa7PG=9<( zv;{ltA@W_3C*%5MrZWZ7H*J0)uVW2~P2FF<(`XGqU{M*`h6pUH`ztY+eU=~Fr1kN& z>Bk)ZUZ)oDy4P|EU@f*KM)C|a!nz3x=C6LA7c=N2K>?wjCv#jj>|iAg^RZ5P80`3o zKlkwN#edW>d2*rhE;@Ii{@Rj@IVnb%3L8_k99aXbm;%%v3(lt_u2~tQ%7cK;M&VA1AHET+t^JdCUBF`3&|cqy`VG-dYiLx8WCnH!MY4@hBwH+-8rD_6qXj+W@3X?`M&pA#_A)RoZa{ z%EfZQgePc6>^=sA1=Mp1^jYI9qf5rHY;sBKyfZb0LrCm6^Gj8oF5B+lbZ_kt|@cQK~n|ayG!jl zq55geT6&sUUWn}5N)XNfbGF{QvVu2vE+58zz-xSt-{87BjLTXD1>(}lT%}X7^6?R; zfK2FM^r@IEALX@7^ss4bRC+2l`}$dV@*ldH8?>y;-j41aDKi7(xv1?_OKzx2i=}EN zxYZTB(PR(wg!$|N4WVYV8@%wnPK_tzf^{zbEmg{#TY32=?8@?)chmhJieibsUNK_1 z*ZQ1%lh$oSOI5k;c@&ws)(W+p?nZCgb;B20I6f_$qzy6BpBqotI)0M<@>mlQw#<~1 zk96nEBuD+xw8}#g>q1xnb_uvAZOl7_lpiP`6)J?&s2#om&5u9#FTtu zMhm^UDrTOnpvE0u<})3>Q0^ojIS}kedjRs7Sk+pKV$eD6O*)?rk%c88?QL%-J#)=ctUF7zuXdw@bV z7Z^wcu@y}q8{sW;N^XgY+nQ`@b)(6Dxg6HLp>#7*)r^=h0p`8{BR+b}#wUOjXk^x5 zAZc(x_uEtDz0H0F|Nq!~%dje=Eo>JB2}L?((Om)yg++G?h@?oDfaD^jB_yO%kOlz( zrKP(|LXa*A>244_^L6k2)vvwx`EjmuUEhC^#d_zQ@0epe&;4j}L}r*Q^enZI;iAHl zNu`ohnf;SkjG4Z(GS5r*JW}YQ2uC6~_+k)iuF1hdU)2ZV6|mM#uy%R^FFsp{u&3R_ z@s?KP)rq>b4=xO@Z0b{ENfRHNe^$0UzYVNK(Zok`VwWQ|O1;wEaT1kjsPx{(prloS7-|*lp)ra7dm^$cOUQnF~*|DAG_PQ8*4l<^`hNi$o zNL*6b10#M=QbQ_`eu!pFl=9TH9q7D81>nm){yb?Yf-qx9tGeBAH27r3NhN#%op#5k zIlyv(5A0?0U~#dbC_C0i$kGS#-)x6n7Y%lclwQFlogOh~6m#Q;9{Ii`J#7Q2F!22mTL(RJsCLy}63Rwu6NE z`*pHReQjdmJB!()sAf8?nPqMG$%lUXoNO6vS2S*tbH6Axl926nwHYh%-mzBaqp|5l zoW`~i2_bn(gS_h!n3@;tW}N>%x-?9suIn203_YL&P_n%}dGX0da{oM(Iek*`wZ`^i zzT_GEW|8(PsZu1k?>&gc!>WI9t72d(=ZIiGz$ZeT=~S|+^%i04FA%bF#UQ6-Q{8!h za^(~y#W!gb00F|E>@8F4FmBv))r8>*Db)MKA#WhuTH&J$@a|$V_v{deH?s6HCkyxj zXw|I!%Y{Y7(}F~o44RMV)tRo~H{#VFIRyMP8Nww$IjTEH3!g6r#Q7-xpzC^Qc5L=p z@-2cai)$I>-9iAJ-)h3&digNN^*QY1;2lC?=4$?|38;`w+k!DjflE9lb}xm1MhX{4 z5HtBfQFz2>AwUv}yDh7;t?xSZR)+G9jDIh0JbwAF==ldZD<>0%8Q5vk8kr7bGVeOf ztVWRX7`>5`b6^aWLO%g1i@tv@qVr3 z_eEmLIWY+j$X1_6*?<^Yg<$Ct8@l)pAGhh+tTitT%TZ%B@lF5sgU!5t14ly)lZn8? zT+>b1?>=vuBC*v59)s}&PmaCzHKw76!s~balD=Jwq=NxvR5^1Ha+t_gg}PVvj#PMv zm#l;QM`|<9T?XY#ipcDEsgBlC4(_%r|7|4=nUKTB@ntV({Tx)#u;LlvGUc$GUqIWE zz?ap5qY~T60qD2n>T(wcJ?YQq^UejzH3z}uBw_UDS4D7||BqdAl^th^{Ild4zO9#G zf;R^>9l{syhqs~CC8Rz185$t38G2?iqrAX>am{Dj4_sz`<{5uC!v?e&0{^iY_KJR@ zC@_)V^nKsr&2VL@kgK(Z9Yc-nF=t_GSGL%`<_gL(ZX@V%fpn(@wu}! z{VeX)og}&+UC+DCUYfBnbEkZNtJ&7E1<>^Gw==WeYuJ+Z3-XxMKNCB`$TAG}nTqhP z(iTAt)&R#QX5v%on@g@#gWBSCn2YQ7lBnih%5R^dQ>SF;{cC`wG=5z(aZ?9HQMzCb zr0{c@Mq<`O7#hA}Q*Aw@40{qq${0Y02-ziG-<`kH~)E@{- zzU5sRyv*bwmf(%){O34djIyX43b?y7Rr;n0+CpDk8nXNu*IF-#f%%rW{gw!o5SQuu zY@!*)8K1NvDJ1p66C)ici6Eq86GAn6H?&*;bZjox-Ef)J0b)UUbVkE^TJb6pY<`D15Le{tEy!?RZzz!`4jpNyT2wuPr*NCFEd!;p@?-+JCCAGcut zFYB1#?Kl^);;krvJ@DI*LYWJ>OFh|Hx`3}iEaQ<{&Cc z^JTP6%)T-|+U4W%Dj||pjiglb?1a5-Nc_}Pq4a^LHKa|NiJYi9JG0bif90#qr!k3g z`MxvIkI-Nm?vRkc=q>QNvjbushdKRxC>g?%UnUWGPt?ypZY?EM@gduQ(D%sYEJxqu z-%P0YBi9k^OO=|d$K8`LND`c&&)sA}Kf^v5hMSP>bCant)gf-!sH29+aFSQL5wau- zeMOHFzyV!X#X?V*Yv_Q+BIi?K7$4d(g+dR-Pr{1@oL!xr=yG^0)n4SMJ=)-sm4>mg z7)rGeGFmx!Eq+i{3$`Adtn1rIuyndzyN(-qkIb0I&ZMIQQ&H>LX?trNe2O-}HM)Zw zFJHXI*z8Lf+TNPy*j5eeVD0(zN?pfkDWj)l*^27tFeM|eHJ+Q;thlWUDBHk%zKumc zVJdFnUbBASJ5o^MIP`53?{u+GaF)OBDqpf5U9g9a5b#=iBVGdPmW{{A(#?rqeDXpL zVrQD^tB8iifAK`Lvo}wYhkVe?kCd{Bs5I?{kKPdszUN@KNc>eR#U#O`>n@ukwbatP z!b6@pva{~Q)BcSc#*o6D{U^ug*IAC?nRUjF*3g@$4ygcACmXGnuun;Qb&f9wi)o5e z$^?jwRmT=pn~9Y(ikZv5j&wXUrXwHWkeitxa4fr{IdFd*;JYO=sgo^vua0CecJipS zM4DvRDlPXy)BJUx1V zbExjD_k-^rO%!6x<1Ly{UOMs}yi{sduwldX%mSsk`($)h4_%e2sC|DQ^eXs=hbUEP z3HX&!>+JRhZTC*vqQ)t@qj7U23)m=g< z568QOI}R_}bu#stzOiJ5{Dk-EYk%!5&$~8vq+F&a=Y7|f^a1lVjEa2AWHFb>?U}y? zl6;Q`+exG+ska6JU#fTre+_S16jEhGvzB2HT9_$RdR|6-%aF{49F>QpFzz31x5CTs z$8{Pzx_VtM?OAA7QsKAuYL+9-;#lk^c14mi97{I%S4tr{#K9dwG7yDaFEs$5c64hik~|HkH9y-9&EQlDoRy`vpX**Ru;U-3F4 z>TV#=o4*%U5VMd-`FBP3TyK!XFvMgrSm?b4V;FggMb?Ew?w=mIhboGPW}4s~Y$9aH zSlR%^=yhnh`-#;JjE{*&PL!Ak zH*{5GzM*YGT4ES`a*M3$N>ZTKU)crYZDkly&7J|Wlh`)RE?a{+#%>4BshsOn%>}-| zLj|bv&W`PM3^C${h#kyt7^F-Q@r1%!np#n1gt;kPAnRca{nC1Fg(a>+8euBD+@$`9M52lfBTK^{>4tBPrJu-sSElB0vkyINJ^&aynKo%*6I zaB@qxvI}x>=Z@cR$&Q_ZseKCFj8%rEV^@(`Q6QxXrr9OOy=CIlb_u_5wQw0SG4Aoq z6~P_pF3Ho}u_4y{)YvWzGLfQ>h+m7TMO|#$!KNIhj7{qMEblR+s;3hTa z+J~#&{TbvrF$xJiud(~g*6!vYA`4t*vnj)W=y9FL-|6l>d>&;B)_mg3F{B#vqw}-N zpziceX|M-1Nzpe@>n-~16`6fv+Ts-?tEIA~J zY1CpBXK{Dvcj20%+0PEU=VSLAl0L$WMv&!Rvj`@Y*Jv(d3hJ_P2LpSW%?~#f+slU% zPr|d)pW+csbPS#RqEbc5JqKO?8x(U_0AMWu}%P0e7vFqscC zqJRk@mpYOODGY?t>L&jb`VevuSDE}q!%Qq7=pshQn{4IwVVh!fNh9R5?PD84u}?MC z)@aCQ)LMg&S`H4ucp$3tue7F+kT_shm))D*Z#D4my>ZyQU6{WP5b>i(Wkc z9cTR%W7`VbjuCu{%D?v_4V@g>S=v#{s(F3oj9-+DcMnhnL|N+{2={O z4TtnS-q69=t?~FxDV0Ir0=JUC7B-UPn#29UXGK16_3p&zv(*zFC zkCSmS4P2hrf1lj=$}kc1xQWg>_vO(}Au&}s#ZsycRX#~|)Rt83tL)|?hH@@AlabA^ zO5KjVrY@OAt!$}3`SypqxBTOu-Ct{*p+r4sHw!y6+4kcUVFOl{5qR?a!8Et|slpcuvKRZFG;iJyg=2)_WOVWg3doay&d9&O z8+>6q`~N3+-qGmy$EQ&JpNyK8?rhC zO-~@1PFF$^t35@=v*8{BsVRzZT}?^0SJ@mr>rz*k=1r^u>dc>|YH0<3gAt+6Gz3QO zVL|-_DbbS@(f$G25dQK{(8jH}$Fnni0JMQ^+^IyqYWol24;vmPM#DG08IFX)-KwYI zKYMT#P8}6F2f=Mll28WY3q)Q?=zM4=eGJcyL`LQ`*yo3OVQZTuN%yd6BqXHrkPx0C zFdXbl`ry4c&j3*(NVcjd4%Et%@58+iB=Z3G+^Z08>NL0$xc(}Y0nuG|xIl&o4Vnjc z?^pfbI(X_f?6=juRy~$kS{_?!KWM0V+7K0+i_mg>DD4-lDhs2+)XD9!G$hIHWNn^) zsLm9bFJ0$!J!4*(z2nm)*BF@sSVS2qdhZ>$^Q7@&j>+EitNj*Hvr9YBQn&c&sH-gI zc6DVtUHli8gL!)=ow`hUbRrKd)yY|ObBx`yroQJ1JnXp#bO8?PqMbW{Z;BO8tNV!j z7Hfq*<4=U#FZ*L6v=q83neD}OZcTsNSPM@7;s@l8Nj(gNxI^OQv+;NI!%VaP{8S?6 z>p66t%Wz3K{jG@s~HVTLZId+TdHdOvXAg9h~I z`<8CN&RCg*8XyeJVsI7$%W&K1i`;*rGY)Y8I-_R}GeLKsk)~Z3LT8}q$5P0Atu=Py$k0eFRVb#3SdFW* zN@*ba7I@-v@2|eX`%ulxC8Y*CcWj z=_3vtLo3Or-+1MeCxiB6b{Ue!*NBOkKCgKao34RK6#KTz=Pga#zn-t|y|5uHx%P@e z7qNTgp-(o0^r_{wO6{tdfT;3cEGV{L&{AIx2Y)}DP5$VZ+U;74`khRvF2=y8d9HHr zubW6vM68Z6-a^)Dj(J9eSyJr5$8Ql$T@{}Pac1|tA9IP#eZRNq{E=zkgYXgHWOT%q zAq_fTN>;cpx`v$q{)O(=(dQ8>flKWlJ~l}h!5?iw=Jcm=pN7z_V7vJD%v|6pP?@g`b$ zPy;5$O7dj)AxG#z*~vRfIPP<*eoEYKEs*k1*6{gs60-S^d(5x+$rfA-Z<%|13yACb zvUi+Syx@M4F*XkgfRR;&CZ6uBhn32^cshE*i5SnTPv4sBkXd{NooFix8$a6heC|{p z)2`@F>YL z18o4Y?Nxc$j64l~53yf(ITL0KtX2ZvxJHdn$(-Q@H)u(~AHC?%aRmVJmk==$l0{dY z<|y6Ar!pg_&8By=v@uU-xc5Yq_QR?y#Kd55)pYOBGJVTWs^=iFiTjL4+UoC^Dvu3- zE>hR8AU$5VCo6d)3x%u|8bgLn@1ftTKC0S-E((jIwV|i|*PwGJZ&XNN`9XqU?%=SW zYsfJ`Q`S8$JNhe^+rFX>gU(Zht=aViK4a=WBlVZ{sYpM9?s$2LA&wIED#MfX*l=ia zr!Gqize92*jcy3XUd7q5x`0~jS}Vvh&fSq?V_?IxPM~>MY=79)>x4S@;d6+Y4&6PW zDK2dT_f@8EfH@L+p)t?IznSYp;1U_lh=miMrDq!PxLjDguv5fV^`Jz-H>`ej@Q#k? zNLtKKBoM|II3L|*{~TRRsYkG0Hni&4o1FTOyU7xGoP9BG2)Z?nd9%bv2oRD9y}mQ}tF?NzJK z6hmW5``dHEVl^c(oG}w%7*t{2UPwRr2vuK9Sgu>3$aswngk_ zoq2ofPdmTQ+O)voG63q@Ja~w;Jf>)e&Es7@(dV~=*!co&wesLBh{?(j!yfdkMhxl49( zY4~fHKi92w6D4Zy<~mWLgZdpf7jhE|iC;&hQsVjeAicw|ux z;$-q%G}Z=CL*on-T@dB&CffhAc_@N?fqZT^-yZ*rx}>+Zk~jSE)PgN|P7pHf!>7R8 z1vNaPJbo7-G?UIR>*Iqs!_Ga?pf+I>5%S!^zWrxaI5KF)OC&M@Sa2qpBg@#zhMCy1 zG;D{0XMdqGNw?8H2(m0%zM(fupJJb*w*Z=)RK6sbiE8lZtvFGH!rur7(2N0;^gU4U z#~$%_F*1GWZ!_at)v3WCAPN`>^)y@0MH|O6QXhN5^XBJDce%qjTea?QrOBbR&$$dm1E0HH;4<}0J34c-tE3f~ zqGyoNK>NY{Zmhsq{#1&{Ofoqnc8fw~akyD!yLH@6*QtAAx<*?ALS=0d!4&w6r;4+= z3uz)S5yH~}vV5O!d3D#S-hDI}hviyKxz>AzmSdh?TV5V6)ZHrY0ebf@5lp;Ei=?6Ud6`w-aWbYWnBU zn*VPF*4NGMYi9vfVefEcGf{?=(yNt7_!5BLWnE9W-c)a3g>i@|U18ls=@EzsYiY~= z4JIhct${XARPQfaIaAntP*SB!u_;%$TgW1`qi!NJ#4!3@AX%?-oV@!Z?t1F8#QN8> z2IR`)yv?N7b+4ON&y@d;P?g4VN|xjU2ZIKCT;ztt7nZ7(bYx)3o(aGs&f!Ntd8Dpzg`k{N6|Ge_@MokrznpFf3ed--Os1i8e(3Lq z^_=xdI-^)WZ)JWbS_Lm3>)kp5SEgeDfT2eMT%ix)Bb9wZKaI(32UmtqVJMVcT%b(+ zKM59V!+**u%DHat$p8erQ8T#eJ+ke$fN=>heMV@nfLEOKki=2Jtv?DzAO14sNuyJq zKYK^W)wvqrA-1{vc*9Mo{dwH=oa}8fo}vH2vLFbRHzp|OuxU4gkL(Ap>B#la=_G>2 z4Xg6Q?BCL$bCvX(Nu5zgt}ccNy~b|ZNf&yA?R0pAj-WBqLlSB_a-V-Tng=1}fS`d# z`MpFOU>B8C{Y3%bf$bx|^!#nkLZr}c5^SX~QaLEC;kQN?d*lmD31PFowGv?})7bdX z%{k-A1zd9u;{r2h5-{eYd&Co}KIl&$eEZC%>t+SbjZK^vK-~FD>o;I~y-d zGSe~kyTVXRE);rgeTUN+v(p_V_%8Nm<&O%!Ih1fd0z@YnibdVYk3V z15z$8{PoidKVC{gBx3nsy4SB+7o*6|)-J7BOGwC}`RoT~kCBpMrg+?9+`7FO7>_?1 zVeTa$TQ*5o08@DHLM18mZ8qATx4kAzGVYPV+*S&%6x$zM2)oxmKq}i&H0$+LH1{~~ z4@P<iG^(N6bWy>VK3J4GJYb{yKD8XtA#^?$&`!MxhS0ZrzLBQs5i$`8cy+FV^h{ zq^XeOF%BvA@T?V>@LG-;-WUXm5bB41oe87Ku!ZBnlvUOEzZWbZ4B+$hE#wx1c+S#=zo9{9nPnUMCF|+a z!u7@K7J=x5Sj9&gjBOkE`KsHEGREPDk!}t2U)IiPHP!@nYnEde9mULdC~~&>_oQ5CdWTu{Aet8Rv zBkCXgi2|uaG?>6&>2yDxAJZlMS_F;)H%{_)(hAjEqZK+&{Nz^%JNIq=ccS5&WU0(u zn&|>_@q^n{zQzvQu3}Yt77GZf38DbjnEkYJyDrh|CNMty@EJ)+p6r0rhVKs4tJc38 zyORUU=ymF$a2lVoT7Z9`@$W`f=`Dl4EymdqX8Tq-7jB4iop-XxCMO>EQS)oOX(6Q!mjiB6wgaPRc zz?L&`vq2Ax>PM|R2>vy?#HXRBSBW)~gXKjy44dut6m@2D#VRzV9F9+76)efNAWTuv zyyM>4BjmKq2g94onLN1 zUmBlw1M#65+m`iu#cM(0n>m!Yq!J6vc=}bqrPcKcnq8v<8?6nBRdh=Dv@vq5n}UJ; z@3#g%?7#OrTgsIdAzPU%;2ga~$SwVbFr4NyHWkW$&3tCzUi+;W6%cx^{(LE{0>q4P z^|J``V`0AqD^{xY#lkx9VaAJqz={(JY~;6aa^CV_YM<93a;5*Fdq0!nwyCKplviB= zHCsFv{V=+A-nFj)jdwg$r!FU6NdV3-M2vr#ieB{t#GaG`;{(p;M!N?ez@!cGHrcr% z|DM$U_5+nSl12k*%~jh#k<+IIcy%@v-}w3=8FHU})cQV^?CpNYSq`j0?>Tg0zt2Sq zhA~XhlLFj_+J5ADyhtio{QvGZ{ZBKg(F(+%6*%N*SBR~w2_xZr;8AM}qdx;R@-`Ur zj0!js6n_ndvdaIX|Jm>VO9TM;ikO4z<$yQgKl{+oWi15yI_UIL{v+DUzrIUH09g?b zm=!7f&wnw1bU|_wyr0vR|IH8oU#iEyzsde_`2D}6dbGgTXljA#wU>6|zagopC$)ft zxqOE`^o7Ti3>ZUWB>x3bMLiN6di}}4>+x6U)3pEQ3H!foic6-_0rTkXl#u)X-tqju z{V)Uf4rHj`;mTtEkIsgFo>1b+;M1Ga`TSQq?f(KU{?qmT|6{-f^G^Z@s{|Q~p){oJ z7&$8PSGX?+l42lh6+}aNZwp?OQX+{F{{<)~4>`&V5qx}nyGNN~z=sjA+q?sze1R7K z-+Og}6#JU>>@aZs18E&VoK}$&CF;TE+$##s6H4$loVqofsU7vV7PRi(0`?1i*NfHn z7P_`yH6e09^~|HI(TtmTPB2Zqf(&&cz>qTqs-F*ki4EIR6&0FVI=MSd#d@b$!l0^~ zdOV>QHwT{h6wvIa5JiE3*vpES-}=%I$=^+K>#~R%qzCB?(VaypOxZZji2Lb8Z$OkZZ5v>-Y=y8KZ#v z<-q4h>j~HnhVu=lN6P$j@xY*>KLz-2ud0nZAy+iyShuRb&khyWAL*-S@Vw(pvpwOW z)^5~)0DmB9q#JneMI(j~a?UF%9*!61X~%OAJYoZ}dc7%HQrdQRYPQyi`E%qJpL$ro4L6RH0BN2{7Wd+UFi*cm@Ll?Dr(rcv z$|M9^%W7z7z=%6~nmWXZFwyV``J^TZu!_Xd+SzW~)hq_Fe4+`X%gzo&A>hNuqJ4uF zT^h3OpBd6e5G_N~9)bgpyv%FuG2(|L3=l!r4x4-Q|{#e!fbY)2sW< zRDa}}V|{)BYt-8NN_JRST2xh&{0#tiTpmHZ4-ehx&7zXsc>tBTI4xaocujpOQ?moK z&#jz`!V3e9a+ zT8#=$i}p%F+MhDprg4RCU#3e{`7*pGboZ#Bw%IX z&yAZKSe!!WG7greZ#H`9qm*z`Cv%L3caBG&x;+)UuC#r{nVkI_Yc zNql=5-`7AWiiam(Aj}?R+Jp+~^S7kKD)@M}@Yiqo*{xe6zZ!TQ$hh5HtknbdA9Hgk zVi{~PW4i=RtgG6JOC9es|9cM|fIByoEAHwF z&@_^-fRTHlDx>xNUnBbsgP$94QDwT-SPHRTmLYh{3zsyUK;Kp{RngRaE8x#m zgXF=^`e~n|-&PaIy3W>)z_no1g-kv^W~w?K#Vd9J=eL$0#tIFtaaT#~p#Q&UuD*&hn`J_EW(P1~<^S zzfzub%2!Gz*gf4sv}FBRnvCmh zizdp-s=L6cxPV(HUX}HVLXjMsCIbwR2W@)yy8*PDV*sn}qvQS9N3`*LZe?y;##!$P{X`czPkMZ#-a4Jiidn5=< zC&5V0O6FtF&J$j!7I(iiUk_LQT{pioC3)obFvh+D+ydgk0_3PfQWe& zL%m(H{-6o|=7~3tYxX`*QWMbixnn47(&66%TiT8wQ#sGJ3*Y~qW4>hl(i!@{MAH1iXE zIPFr)f`Vm#j~6t_^z6nGO84*qZ?dx!#@Bb5C1Ce!aMSO#YR&SiReb3g8OK%ca4zW%+SF>;`M_T&1> z)|rx~0%fUE=unW!`kE3bE$ z1ti~J!`~!XmuYmq-p}$ewy-49A3e2eOldoB6!!9W`~>H(qaG6&+|H4AN{Ka z{rW*qH0@~}v+I+IGnGQKC~0P9%%C&uE9M~)9ETF!1-tdDumJBWioHA$Ax;@_H0exM zKWK6;S^>X)*|m7M0<2mge~K%bLmP7xu41(LZArlqD01tz?y&JlcPSwR5qe81bCw(B z4+RV!h|+#G1j5YuCOI0{$zY>}B@V}Ez0{1zO{Uvcw;iLEoHnM$eS|npuRPErZ{a_X zLc8-I029YXJ|=(xojP1YX{yTuM*Aivq%~b5zJg2d59bIP*=L4Px{)sIS8&aAER{Iu zu<*_!Lzjd?;%78#7_1*WTV+8iVc_AS8-PPxv@S)m>E@)wv;8yimG}HE3ni%4jMiEJ!Y$p)MQ-9bcSe7SZOzHq*DB`KrdcG z-|Mk^-lP_xV1_d$G#Yc!hg~#w0a~0$;Ki0n3aY(riw?# ze(yg(Z2JP}&C#`zP5m*mI!-Kt9vpq+Q_^re`BfrUIp8ohQ%zXF&A4+7caDQe(D<909k71$E`7L;HS6=) z`t$Z)CRlHP>ISS&={HS$3*8%|`B&16X^|Gb`kr#5Kw2^99F7?xzK7{`7?*MEN=HD3 zkF6fxf;ofh!F6jb%8Yjw`-}$(>XqF_wO+bmcCcQziKkK>r2u`y=zKl*_{zhT>ye!q zeKG=>OfTFTH!TYR*_V*}?RPT`5Rq^m4k?;eYZpObbC zC1#S0lF1x4*yCNkB|y&gw`^e$pQ*JRp{oW+w~NI)Ja7+LFDJkVlipv+9R1TQkJgMRDuzT`4&|mRSnGya;`)W$YP2&*0 z$E~u72#mpx#a097ck=u>2nunDV^Vg+fH}M&oU(Imjdc;qC=!A9QX2z3geBGW}IGh>+gtQ!DG^Cfwa#7>>@< z6J(MWg@r1Z8@sNEY%?f%MjU(yM=|c=5sJF`9&Xd8Id%|jcgIusdaJ;xe&kp1>We%Bn|egv@_E8{u~KS62FZHK zD}sBY7QD6#;BC1^N%_t0<=`$tYYCfNXgsc#tPvx{AXO%|&L~BeG{?tGwP_$V217l%Qo!i^%C~G2>7`1hsq0=?IW#x!99HPT!0^uw&$k_VH(J`TYbTO zm+y}AqUt|t5U%4N`%Qht{NDWi26Eld#}y~P{gP|Yx?0VA?5o=&TQN145kBxjzQJ!n zf9$&b+*g9Y*@&<&4g}hE3mN7dy`YO~Fb!^Z1;Co%{Evd~#Dzc&U3Ij6!Nt2W`8H7l zjFH$ICcp`k=tIdT9n&>PYJ-d6!ClD#$~T+DaFp95wuALDc|FyQJGKG457rdWtzyb$ zjhrrNU#x-x$JWz+=Qdxi5@L7)oQ{`~xAekmu@y;V#-58*58{Ah#=Dnu~MyULt~Wk@-mFLM9p!L*OP z#L-4ZR}Q=}d%|q(p;Oj+qA{M`0LNT0Zb`!2hZJE2%yD00K1V2xDpZ}QdH~kDS2xQ2 zfYVAtL}ri)P70Vgq`3XYJu9d++CP{c!gs95i0{>&~0h z!y+@yh8V)s*{;mCT8!?jp~7FZW;EdsHnRSl5(bLhhf%(pu&s9dKnZR$#1rw?!YcF) z_GU!O_9?8XuQQT6tx2=-)ex#XVI$40;OIhl|+j;CR-n`67%gN*Ez7FBskX92bf%bA9U+Jv+W1i z#8!*2U)R#X9aPq8`Ay0v*Ljz0f&^HUm`0c26x~a6QFy@P^ZWIJcBvx8@G?>t2d1RaW*ow}lUfI-hZV06UXr*Wi`gMf3VP6{)k&qd|_BEKpTCcZ;b6yNVYx(AW~1yeeC3J2#wi3ZS*Q zu~4m4)H4u^kq_YIiL8ihFeqKM3$QR2I8H_3$2I?W{ccCaoV2k7v=372ff;iSv}yK$ z0iW8C=pE@v2>pVPzTDN{QV7cr>@w#X$nv{jjE!O4wl#e`=aT1rzklR!X?z?f9@8}b zkqSr&nHzZQ4p({uuw9U~=I9_usDFv{rBvIbtn8w&vDoE99OPKo_9|xch!A#<6r;??)%VK)tm4_wvGFWjeGY(QWSRALQ4Hg|7 zoTgkt1SHWgZJV*LrjtWRriKc#ZhRPHa&a8C(OfK>Z}VJTULUw4kR_vSR&J%qPsI-? zaVfjX1llv%zMP~ATzzDF^o0Y)>JZ`uu5K?XPPE%BbZw$*dA>Tq&PuRwOuh-*gQJ*fa#+ za(g-U%4Ci6%8q2N`SkA@t9hGG3UBRa%hNw8_7vA$!dDXWj|(S-4*QHe6HaqJ!FC8i zU2y+a6ETq8MF1@ma|0(&6M^5;bpym5II-LMr-|MXynNZN47}pjvmWPSP-6VRaMg9k zp!O=Qri9-_5*o9N-4}^>D%0<5okwe0BnmZ=3@y#b*W3YJ4{Rx#TrFY~{e zK8Gaj&`oYWhIe*QFfwa$59*yzqfk!U^nQ%`M#!wzap~Ljj`bU|2KM+>{6fheX(I*# z=th}Q=N+1q;W1( zBbR*^lJ_)YmzV;*H)NcXo{yMX@lKE)`fL0?(-$=8E2<=9W*qEk^V-;Ezx9($3DCKS zc$=`y7HRJWvi@3X{#iaTf&6=LB);(ueg~JI&XkDBiBi^8tU`ll$%FGy+6c)!tYBEA z>)U7kLN1Usljk+DLc%RnCpgXKStqICaWV(bBlB64Er@_R1W8^X=F3Y<`J}1Pg}HIJk+_kAkz^ijllAb< zauj6AF=fx9mH+6iK@@2<-}82AyMBMfEZM=fJ1Iy=!W$4HyxkDu$Q*_ra|K<|cMhMV zny-68NqVB}e;D8EE%hv6l9ux?$=^2Q<)}{6+i!qQ5Y=C3wXKU7_4CElSkGOQSga*aA5nlUpq%ZvjeOkE&Kc45!6 zRY;MhAXYiB8jO-@bq6TqqCB%YR|6jz~-Vceqk=Xe0{D5Pc5@cvegY8 zTJOxuMl{Lr(d^={%~h16;R@FS7fZWx3LocZwcL~Py_dO?E~U_)w#2Dc4kQ1Wsz@D#BAF*lUqphc1whv+3$i52aClIV(JuL#LY|tr(LH9x zj^~b|lH;n=K8faN%%t-|^T~joEaLMn1HP6V&<}10JEDIshHwSnUM|`wT?q4I&t|Fm z<}tLrPUSbZdhL~b0``x2-`L-(B$^JdDxu)o=E>JB_wH#|^yVc0I(f{#+z9@eS8;p` z#qBLr3y!drUhEzerdWU&6x|AgM=+zNl@0sw2Hg70V$si8eHgaJ;)Jvk{21okc2|{_L4W@@#y{X2v15O*X_*0aoeM1|H8?&tp>S+ zlcj}3zqqR|s|3%7v1}(_9Shg5el8ekdY^zPj2qf4J;cM*ol^cn_3OA#X@aHbP-WQ$RFZJ^^R_gbaF;OlkB(1bly*yGW#1oF`N zmtkh#2eTeKt2=C_)%4Nq@xpTLV2IN61^3Oe35%EMILwX?`xMrn~miP#| z-yRV@ZjaL+N6*=ItA2moWMFW;MfT|BQ2Nd zg)tD8j*TA7BR!p&nP43>rxFmL30^J@(qEGndqTg6Sc&AV42l)p?e4)@u zbNuejeMMsP=#JK43371B-emox2fi#HyieIBKfd7NU}3Ywxg3BlKg+6-mTSu!@dv4` z>$)*Kp*9<6V|Wtb>aa_C)J8kmd5lYGgx&M;=>6ZXDN#8l z+G+zfibErx>hp`@20V88d5x9tgBwL}6tV7|dhc9FAtCID3#S05*!k+T3PSulBf_Mio12v!(a@( z*9v{A|6&0=lq$WW627jS)ehT1^$5B{Oeh|H6A&$ugxw*ZnjCA3{qcJ`Ey5w1GxzTs z#a4)fDU5L-ntX5BEsam?VLo`5?1Fd6sL!%ZyOX$f1h4xH&q2AO_s(oQP7;6m zu(YB1n`DkzcO9kLr#5wH2S?Qs+&+}7S9as9x@+J`hKN($x->?VAKmBG_iv(aO$6SK zTzb`8tyXJ*d^^D2WzS{3*Teo%1@sFOtHQvePeiH1=N&T=`nc!k{XLL}<#tiBKp;xi za8nPDdt^{mTC%fA2L50UwPbT}lhk@h4H#|KfU(MzILSSL!i}bR2tz%1-0CBf8Mec$ zK$=Wi7HZ7=wFzUQm#~PMchV>Y-Wt^RIeyY`C&lW@1`D?aZ~LwHxJ{vYp3;GOfa!bs zNg+lsJT0rf6O0qPo^cM(lv1}7_C~nYKoiU+MC$We?NH2c>AxF7Jxc7{F?zrkh0G_U zs{=K?fZy`@c?}_Fcp2xA#KUIk9r8gmNhzahN`_=OiWEo80|aX%Id2y(Rh-USiRVO2 zc}G&n{y4ThYcjuJ|4wnY-?rTJJntx{zfrri{-}+&%R?8SDNMtSMZ>16^fBzb8gy^nC$)^nOLi&clx5?m2 zFZf1lY*OEZq%)f`3EtOfB&0ys!+l*g~HB1&=={uZ1G;y+v)l`baLkgsT*oLR7G z;3qJZ;K{lk8w<7SZ);vDq7PCY?D$%?vjOzReq_NK=8HTLntMDa*$)vaCi%QG0y=TX z4t^et$J4mz_~harVdbOG$1MjVTquOZ^b(Ph*3&`R0WB%5TE}G5cI9>>FGuons$qaa zZL3nG)h%e-lt%HeNDwye9@VxrCFrV+o<2gs{4{+VTnWWMZ~5~J!@EV{wul#!c@kQ9W0V^j#=Yx zhp_vcIrO)fkJ;5E)o_0+v=uadyWk#p**i9>BoX0l1-ut-P`CmGzg=oQFb;Z`t?R?0 z^zTtVG|zumlA*naMacBJ-U@Nz=JhZ#`#1y7%gCWzy3`LT#=zM+m>3%$z|j3u%xO?y z?X|@d|CkNKT=-ai-xwc087s6p*O)1NmA>_i7QF&JPxrHD-I@6BJ85uHA3ai3q9w^k zkBpkwV+?y^j{CEi%aBP_sv%F--*UX_4b1OCNG*josnS?M|ma7!2-bp=gC zOoYvAqg#ZwNICd0th#Zq$dZNP6a7D&y>(R7ZTs$xfP~U0f^>uQpmaA<0wN&YAl=Q&|g z&WKQv$B3Zn$YS|~$htQmI+oXQorf&|^uCeMl3L(4^%Fp+i%aX3yX0oNtb=v`9oKBw}O@tmxJxW%QN7fVDY;h&yl*$LUp zcXS}xKsRmI4;w8qiKw!CFIS}zJA&AaMDonOiakfEPoL{rcI(I354w&>TrZ+fLy->= zZ;kO@v{9mfcGC?^TtB|HNZMYW2aHVkljD9(OCn-7+&Yc7TCHIAZY@O4QNPpvaY$N~ zl=>E$FUs5}rA#A#!Lx%@_ie%i_h!P`l=T+`=^S=_GP0ng`5H`M|J#a=em9wxs8Fj< zGK%XIZTD_~&ss>6ny8|)>j#2noR4m^sz6!nugt5NsDKE6zN6IFi-rP~eB(UOD5Linu{fpDHl zL1EmN;es%!Gd?K8xj#KYvxxpNH3zAhSvPB^>@zGB?w%xcuT-Xl`HneNP z!SA(6Pngp;D8hW|fhoa5t5gSH5PuQXiy~Q+;*_*%wdRq3jM!a8>F1$zNR;P)asnH| z=L~5PKI?>GBas#=G}M6zDC#OI)F8g9@*GWDcn<3dv#MpBa#ym;-=&dI(e}AmgId5| zrLVLQ4R*h}0*2r|CanAtRwnm-rWo@4U8z@9E1^EMap+y?(W&y^#We3aNep-Sz&>jV zk6}n`g?3us2N`Y@5x7)>ik$@Q*QTY~)}TIjJ!x0TFh`{okuxfW${J?|&FD*r2_&W~ z)Wo@c$}G73T)S#_X(^O7qFSHFe5ew(0?G|_SyCwxfv{=9@!`DQCF#x=iX?RNW3Kk* z`Q##tDSjX0DHFCKGJ(rDNYvKEME8wE3#IEegzSqetU|y#S0s>+Nlki>9>^eNk(iTr ze=v$Ct%61ice++ZESun-mgMLgtpZsMhf=%LiPtI1wU%Yyy3PXKk`x7hkBRQ5ky`$4 ztw`_FHteqKBzhErUm(9i8hYQz?T@ab{AkPKIo~DA-_FI|Dv!#0x@e7jZE^w|v6g&e zacq&cdBtJ!-1@=vi6Ic#Y>1IO-c+JWWc&^_8IJuN1A97kI?QR_UBrZxfI*U*)hXuL zwM4(cfrQaUVXVEZJ>z*2GxliCp4TLOIJG7R*?s9mNgoL*qiAzS{z+J%+E(qwq2N)2 z)~S2n)(M$Ch6mcIW`L{PdLxrSf7Q>@%XRolA0IXizZtPeWJ2^hpI!1%3CN7wc>Z>M zUOu7d@40S?7!k*ziZtCjP(<(igYT}WT1*Dty~E^)Jd&xUz3G&daxv}%4Y>6RFI51% zgwkpP->H|qf=R6lz-RZ6>a57IV#oYN-T|aZZdb4*0;X(C1trIxWTNY(G>`qtKv+gx zQG35k@d7f*?AKc$ry^~GFy{FB+eBb)gUbEy>YL$~c2PXrvYW?+kLprAMn6xs4{ ztdLoy?i`Rs1hTO7FzszT7;??rsNeWq^*Nyf0SqlyCB0fG?}i5fpaw*zZ&zz zmR(F3D2#k4B3WhutXxLLCN5UYKcD@;|t zIX~kbU)S=JEOcDG#z0u-4SL@V$#z?C-}rLhC+O;UqOvIF8y;$SJifVC=H@*s;oCpQ zj?UletznW+bbg^Spbt8*{`!rvqg(_sZda8ZoO6Oea!uS6!`R-xRqFnm&TboEVa<8<3bx>#fG?({G&y*H2!}@w5{__cxl@9SBP*ew|F^E`C7?lv+I#DZU~HoweZS582na@Q zZe4FIBa5W$qJxN49X-q>OsLi0?>_9ZW5ExU%&Sj`|+n>ZJK3a zpxR!8b4K!l4JL+sYdgj5!BsPj`Z*FHzPr+>LIC>zd*n|X1k`P# zS4RraxOIg=ig3Xbf@T(Hv)MqnhZfNxq8Wb}y&XL5-!mwIQlP$WzZ*g5zFd>F_gJ)z4V8l5p(WKj6N|PPmfIxocAcNdyh~FkS)g4aZfdQtiQI$P z#>P`fvy^VyQ0qKUfUb1R`Z%)gvR$GgrK9nw7TWR=`CDA|k~jk4+BX0{s7cT|S#5vF z=9HxE5axO#p}tLHk}^!=-;5#7tlVX6hcqYiwt_i{MRv_$|6B}DGVAZsntDCagz2{FJs z#h3vPmoNqy^4^Go(+b--L-samF?IpBM;Ed_W^=tgpKzn;;Kni$^0Es1N@QZ`+&86- zUNe$ed`aZA@~qpKF(R(pIMBP)fk73B8;x|S9bQ`cj-I_xm3TY(10%E4!J4*zZuQKpuMfdF`Ytyw zM?1LAiyTC;a$t*5NP$#rWO`n%nui@zy*h;ILf~@NKdx?(w&J)+gHiQ`nH)xgNOc-d>H+R_|c>c$#8ZS!1hL zZ4zJKFsJg&r8i`nd?4(BTwXXr=xcx3&kws+W2oLd)h>xsOvhZRxr}>Az$`PtxC5W} z5?5*zOib;mn_?ulP1c?;P-INun0vAV>9F}~NzvZ}s~asdBRZcwJ*0lfcYNg@MZl^( zRV(~(bS8v+9FGg<6WrPiY4mo&?$A!^j!BEz0@)NmZsDQa7OoC zyh-0lHwiEUT0ge>xnB1;{+hukYa_j2KW5Vph5^HJ{wVLAwsrMJeTPbIr#c0x%vTq- zkfTVnz~sxXu$NT@PVr^uFRbtoNdcX`St>30{&Sk}dQd$4;{$E0%a#H1T!xP3UWM7H z+)MK;kSy(^XCs-Qb2YLJ1QXl9l3;C;0i|fdiOdb&vj%WpJ1pa_JQWG zJ;U{nc_~6fQ(Z11{JpVdH+4GPtVB=~-pS%trdx?~Ofe0D{WM9`?2<;03!bL2MD{ur zqQdBkB;Lr2w9ZNdVfQ{l(G@NgZzw7r zFQc;{g9VU3n@YZ0$pq1#*)`3E6;huUsp*y8ZJs*x_+E3yns28P+Rn@F*88?+;n3tD?vnl2rCGeS^~GS-`|$$$5utb1o@(mzC=_KC4V znER0_IBSBHAjHqEyZ~4~xr(rC07;!18Wkbn*o;~rC#-}xw z>q61ys40m(UxVy2;Pcb@3=fp5(qdTr7*|q0Sd1!gTayM^`J9Abn~T@+n(4zj-Asyw5aao(oz{=YEK`Fqxm z8qf4_YaGG*)1W)2y>3NFlG5^)OHOZ|>rqDV=M)$Pm{1WsWN2;8_N9-6&HIf>W zd({$_Sc4u!UxkV0l|c8TcT~?@7)ydn8gqi=(BM`=9kZ6+=IEGrqnhOCu(9D&S0_bpvO7XguAiRB z_d`PmzrX0Iky#A&L8y0OIL8~v;GeW@FOv}zt7ML=#cWVBm#f)z?{DkrGNkYbT0Y6j6zvE!llh*k|L&Bx0$Fq7}=zHBK^ za%D5YeCn0Wv2bW0MVP*d^g|ybknA}F#yQ7FbSVgE9oPYhyr(#$tKu`82qn$@GVb82 zNXB(6-e1vgLL#l@yW=E3d*(gb{F?HN^$|*tvII_q1;bM|_g=X_{M`uDv8j;S&8rXh z;^P+>#Pgp>E4PV24W6_S0bbep{?z`M?_J4_7SV*1#{NSaZtVV}#hwIp-DwkL*M+u# zeL_V-jLRh&#ybJ@_xB?B-jVm*)3?}JIUZM_cU3cA#omTtb=OtMB%dPF*9tK5>$hGY z$1PVV3dioV*0UqkAYbc0r&$W#zeFLnVuv|st~Hlj2V#T&v!sgw$9AVjcVFn(Z5a-%;F=Hdx(li+-!zLw1-zgE7C!%*qK z4Ah4u>#t$aOVMY#v4shBQO zB$?f9(d&!9JK=N?oo9Xz=a63!xRkv*&%{UeqZY8Hyyik#fDM^UKfKPt__^GecxeOL z5FLC$+n^+;@Z`aemhbiNra+^G=Uyao{0DG(Vx5i)t0*x{hvzsc;hCa2&~k41uhO># z36P(>$b86$&ZSTt7)2&HgPPen^iejA9Kv7;ZbTq;@<3J$L^}(OGmgoQ%8mLZJAKJD za`w$d$2-B_BR$Bj$~ZfP@##p$u;wRxUtb=o)gEp2;>I-R@|@$pH~JEoA$fFjZ|>Nw{p!OoHiR1Rq4-DL(p-BD&f7ara9{tT-`>H z-pNZ3zZHW~$P%T82tti;N!Y&8eg3K*)chTw{5V>`J@Q}G2OoLmU8a7%!57Huc8rf1 z$D4LU?CvS^bITC(EIa4W?Xa@{zS9U z?-3*9dC!A8C$bC` z%j?mmp^W&rXgdg@ln5|#>f>aFcObP1N?@ifeDIGJbtXY%$6H7DH~|Wgxu>DhlQeml znqw@8{LZX{sM>)dR5vlDg3vz2jP6>#OV*f0?KuC#BGtlrAluz|ENsrdiw`DIu`aWl z2OLIQsQ|W6iK|S-?xDk1_Lb>4^a!6A-UkkB0`yb^91?NM&z0D_j`m`fduE&>J_FcK zm@m0uAxgj_j=jXkh+{p&wgI?FA`iB_f$-?pY5C5%ovd(}l_4RNC|=x7Vo8U_5}i+e z+DEnS)~Cbc-Ca#{3+K~J*{VFq(6DxyS(s$ptU=Ej^SwS2Bs5vycE%utw6Q5{-~1)e zg%%xw>Ia^Hyk`(5R<~%tURM-(*0VBe;(NHrNW%*wbaq0D$J6nS*4@sR%H!yz?}*<` z7B*uLa58)-GiuMKowly9{0Z&;S@WarsE?0i=1^@hZ}EE1^(Ld{n4iVIYS+kX0HVwm@$;g|c75LZm`zGaf&lY_PSTt*=K zuOKKcY11Cg_^B8Qj!YHG#NVJG{ zI*Fe&5V`_ntw-0$R#M0_E1!uTE^;Ckpf~;~sDrvbx!HSdOTxCPH^Rp0NpL@x8c*Q; zqFm+yCfz8_G}R`y9k%gD8uP4adV7yf#(0s-S1|QugkGOATL@6QlG8|6G)N#zLGFxU zo83~LLgr2JK6>S|XYCF z-WkaBoHvJ&Y)1UsQfZ4tiibCJlU5Ek+EK+V+qF3Y{5$hC7B#abnc?H5vT3#^W14Tg zqqem3v^Ih+^~Mi_=q{fP#cu>cRV1HmPk5KyW%01b_wc5N7ioqAD1x3{Z|RQ9u6^)B zi_k_iHtv*--3gtl3Jw>D{h<=nEF+?kuvu#VndTF;)y=d`VwgLH5@Dm~^}4)~mz{33(gfa!`A4tTSSi0>uaK2f#Esp2TdsF)=K=U; zjhkb%FJCsUef}-nEg}Mn?W73Ko~|vV1XtcoHpeRN@N4@Yy(;qYDEj{dq2qhiE`V9% zmyGSZx&ZtFT3-#Ivpd2sy8t}4Hk%)lYUkgA=Mql=L_c`Z_3kqN`I%KJn*Zzvf=^dL zgYO2Smw7;b+O>mrUfBORKeP$Ni3e9stW8&zQT%^g|^V6wA zmNd6x|J|F5i-Sce-l3xo-@E_n1*3$YU^rzT;05r1^&)%<{(U4*_v(xP;imha9V&ti zkH|yp)N20iJG2giPQkBXyGb>83He{WJmB(d!Jv}dfBg=@!G0t(A;Ar0ckZKq^MwwJ z!i9hh0x{RDUtr?gZ-0VbY#Mh zt!wHr%zwNN{`=fJ!M90-KLS*v|7tbh_!}s}@38~4{l;U(yWjxG<{IEJhhS+NJmTPX zKQeax6(d~_T;AD*lPb&q=4=EHrQ++~jxzFK3_iW$&Nm}tFD;&z3&2?75F63xwLRc=(S$Dfla`cyf0iw&wQK_IASMI|#>W7w2zA_P3oIZk{Rq;}1%J ziYR>1t275#sY^g4%12f0z`WY*0I{TX1>i~;AC z$DV?%hw4a9;ORB!p}N39FY_SUM%f(=j*(ts^Suaf0JZWtxcVp4>qxfT?Q`Q%pA8W8 z?~RxHI9pJ_F&hoQ?D&#=$Y0B<^8HoR?RiwKZwwNqjH zUDODqYFD-g!Z(N5-Dj@w61q`W#HLg6Dg1(JjmCu)G!rZ!U&7dY4j-PKuSPY%r}nz^ zv$Zs8IghK&gBclHH`j}|f&w+x86w|+;CJ>=K{CW^yhL}crO?)Mc6Kgp@y~j#5&yM_ zuh;o<$UNZatS^A<`3c#ez1MM;+X{L&ce-VcV>2MV{t(<3zC9M654iu(;cONP7s)lv z`OF>xXJH#I5N@Z`<_@$p9~7|>fb;H?Yn?R)#aw+)Zuw4tDy1?K<-o&waj4J}YpwQ@!`Q+_#Asn1<@?bIKo~cQ1m(r@35)7v3zGwB?$!6F-YhYNRYi_0*I3^HN!wpv)Jp zNL+(G%P?8U0tW}p5ig#1iFW66wO)d5WMGR@S=4UT=hp^foi18QF@i~ar_*$NqVp|Z zs^+%ICV%Y9&hxz`##K0+ zhDWRGs87B5XVY2JzB8N{)yQ;q2Kbx5Y+9~eTC(84lUUEsF7Ts81RoJw=serj2wk_e z$P5^L>_NXn%OD2cK)(XR?BHio^BAeylE06i;2c@`=nxpnx=Mu=N(bfx_?OnQ@cK-A z))c~_|BdO}&A+gWC3LQT2G~4}y8=Fc00qYtOt4T#wT|NU0)(c+!Nld$lCt`~4EObv zWxAF@Z!#c~{T6t{?r~>YZ*)Tn#V(cndo~!2NLeQW$BkUjs3S}vxDO% zW4{F3m*aiVuJbg^!|Ol;rw+8(zYWtJ0Fu@M$FSNPG@CzTg+k+eUY{27-rTRR`5EMy zRGe)H4*;?wMKYQ-weGb=(Jj84v35dFpEabUaPNvBlI?bfe!ubTV=VmoMAwZ{XuK3q zHTK{Ggdw3{i9M9v+KOFXkgNhK=_V6{baj4R<2^n281>Gr|cJCz&#yy>&D z#gJml2J7}uWk9q*Z+=ua5#|nf*?%&h7lqu(`4^D?YnC{q<6(Uqq+#Y&7e@B8GM5kj zucAjjZ!pnD;{~_#9H72TAMybWiL2|JeF!$5Se?}AqOY%Qov}7)lk@)zEyKgp_VGvb z$ejq+h1|HJwKC?j$Fd*k6)*+d;Z_(-22EP&e^H~#=J7iu5tfG1Q4%x8jHnhW{vLrD zw|)b&Hof!Dz$AHyZS=13t01b38UQ#drGABi8|K$Gd;yFQ)#J;n_26px0 zK+Vbsj(8!_it}PrR@7Bb*PIR10n>q&cMZ?|SP%Ak`N-avhJl^`4wWys? zW7iKE_jmu0-yDQMePIP&9lh*5+$4edyzb50I{Qa~mo8A25z7{^V=ovEa@$;IaGebC z1AM^QQhssj&m5#3_c`I@w_UD?(1AbpSD>z$Q?y_q(&&ysq-0Xj&mibDvX4ks$UPzVwt;{6aL#Iyfk5Fb=hvV2>1&vV?Oc4p?aC)O5z@O@&22Z824 zxtVg+&$JY-2F`FP(fZ~aLH;rtoXwPN8LfsZ&2mCD z^T;U(sOxQBji+r%8u+X2zEjup{^I&JUyxQNXc~98s0TUor~B~a@Q4OpeiJB~U7aSi z`t0Ffzjr`m4l`4_1jC3y$gsy}t2zf%)pJvoM%R21J*f)AD_}ou??x;OPV+en0IJ0t z7;(4F9b*wvANw)x2$)472NB25_-KZrMaTxfn3+t(qkT?sc`<^%V-{mGmZ#?Y4_>L+j!vITVA}_cpEsc+hcpf4Gh! zZ#bb$;38TRqH!;u!$&^~`Gqi+qgRE!Z*wf`Z52>_2p=w7M*T6PpV!#(msV!q*U;6a zZyTB0*c+GfzX?BJ-ux`=U@VWYQ%EN#q6nMC+>66V#i(7<7P)1c$lGtZ^$Ed(#WQ+aowC|aE{bh)21byqK z^u1TZw}tS1S?=w^D)1Erli%yAHW!Dxt??7$zXjH>HYog#hB%MHJ&sM(2fa!@KT)aj z3h{Ve1F(!IJ{QZ*>nP)tx&Ez>O(ovdv1=655m|!-{yFgaw2u!GO-y8RG!>j>n3`7=6*DK@lVl(a-CtUBhC3!=)8yv19R`5;dHUaWoA> z`Jjupe7^{nq}mMhqphANTA*}LA^kLtGJcTG?ii^PBC1EV%ag>gpL?m%30c$6?$DIQ zij`7FXsNsL?ry{g+YGtv$ci_|GVCP3I?|$6R}b^w!JiF}|>V;cvh-PN|7y zBu+W{j#EP+8|P}1%^A;;>(^Jnp|g(8IRg1ZTGrrG`ZjEDn7ELoiU!>%Ph6?_&wSZE zKO`(P*R`@)KTaqLk9cZaMurH1cmc)6Oc8TzfppyCnA5@tocD@)_-&PJ7zI_wySghy zd1OQsl@B@GTUg@Mcc%Cr?JMh&rdgHuH}$@7 zE`#TS`S6ALv9HS7xn4Zo{hd^7kYETY9hnZ_P%MzD%Ksx`e>B9A;5a zwi=TnRr9JJ@yFXCq-+~h#0MJ7$ZUf>F;5l!&g{n>O9qQqlE@y82<&z$b`O-CuSZRw zhNn6v=uWD756i=hs6EI=rB@zh;yI1ne@MWC;k#bbZOA%=>NikMYh zgs?IJ&!kXJVOJ1{8@T}#&{qHZl!jEM&!vSl3LVYk#!eT@q2!wo=5V^&CL4bB5aszZ zH2SwBcDM!o@{;SqAoW@}3^YnUir4Re7+y1y7^9s$$dC_N5qz*6L5--{&?C{c$Tbu# zUm~wcZaR!HBwISUmibKXKTUT3dhhC7%ZckZLqNR~gPEe~*M7>;I04W8k#NkZd-Bbv z!%{TgyP#hqhE$Jnr!3)+d|hb`q4QAIr~?sh{~vjNG5(aFk?lv;^Xa#MO36*GZ$4`+ z>nH4Dg(!iR$xay>m@P50^V&rQ%1KbfJhh&znJ+x^rJXjVg(vAU+x4SU@WHuV6ILEw z?QDHSD(p(ZDWKlhlF|sm9MhZ8Lb=oW$D94J;3ZeeN{I|xGrd|_xe>Uo69o`y7$|ws zcJX&jh-AIP`y%$aM#THZ72?|9BQ%=)?U6rfH^>`;g4Xp%N5u&t{w6 z9Zd!&YvHsZE|A85dCU`@%9w{Oks)TFe=4+%+2C%Ktk}gRd4GHnh%E%009xLxb7X!_ z+1!d4`&%0O2lua&TkQkthAS_)@bx88JOZkoUunt6ZBxg9n%&i(F{8kTvn!6aL!O;;j#_OjkDC(VuMwcS8JJuHUQQMltz=#$=+(r6`8>rYHA@{*7 zowt6&HG#ydD->yPjjvCAN<4M0PRNd2l<>`Yc7znhQ~lQ+8g6s;-4`7kpJT2+F~rdf zCc=71(thSm4>^UOD5$Dh#eKx-Fi76SMbhK(E*PIrjCX{iSLjuoX5%#`$Vde3&oREA za;(u#O7iJ&3JBZI2V}9k1P4zQ>}r!rxjSRjZ8S+Qa}+sG9b;6WE`26eyqC(iP#eQ> z+eK`9f>luI@6513UGl^^*M^Q~L!U(2F8-ITDNsMv%||9Vf8%bhcmU#6!b9Y~lOG=_ zUR=L@rZvV9cqf6g#8G)y(Av(vQ)KI&VA_Xf`;u$BDaD@#x zpZOQ+YopngjBY&dL*T1%0fTXxX3es^|FEKXf74B$=NBkv&C{K+2`7B-tW?T9a(3(# zP~dAR&!#ltw!lv_W=+UI&(KGLrgHY$RBLo!HpLPKCnM3u`TfA@oX6!k>*8)6nSFRp z@Z?JzWPxwG zlfxE(&uHj<-{`x%K?&fPtS09p(Tq(KRs5=hj8WMC}auEGjX|(UcUZjl=cjJ4LMNbF=bVte3AqL7+k&} z1fh0rCQ?#1e}#q1D^nn2kp`mMaztH_IWf<_$EXypV@6-Y`I)wm1Gm4M~>w z<#}r8#u=EM(ft#Go#Tj?ZZw@SoRt!bw}BF2WQ+v;?xac;2``dl3F1>zVp-%#N1dvQ zc44`#i@f^vT&asBw`0{xMLTPgo?|K}M3oIAFv?gl_2ix%4sDvrxIROb{@XXqxt$%y zAYO1L{GPtLD7mlfW%Xwkk{s+Eah`n6J=vjrzgyg`l{e9+MOvFYkG0KoO0j23vP*fv zzUsZPTxZ35tE=dUe+bpN|bg1prkwSG!JeZ$;!EZ9V#jg zIghzdPtf1f^9>_7mT}G)wB&@KHyr!_WdT_7co6Mw!kNdws+{lLuoUo+L{)y(_=7G1 zN{+^xmxzcAPe{WtAUv=(``_d0XP1WWqT`B&Dtl)=-;?!bvrlRpynpP4@-)*#z~@zx za=BGq2I1p-1v+d-SqcdwvhduktM$b`_8O<(b>>-PH$OS!BCqLY%-bl^a9$h_Y^J->+@_>*m!I!>QOz2)w54L^x)+LA9uM7rpg5HCi1(=>zF1a=Cb15YCu1!WBy%-BOm~2T zs|(t62$G>XY&x+gT*M5Mt6n9u0L{K-#)!V{-xOV+-M{KLCHrAZf&Kiy7#Z6Qnpx_> zoGJcF>RkS~7wODV<)EbfHfbd-WPTE2Vo3d+AVMM0q8I%WRmB6!PXHv7UoN#6FmN~( z(+iWm4N7s;hsVeJ2p>o_s1T_W?l#shqd+9mKmWNp04RVuZ`TAJ=WsxTRP5Q-SS#C= zos`7Ahn#a2O~p%7Zc_#!7gc@oav@aWeKu5+nubmio2_WyTh47`F0)T?SpHNcH4PBQ zDDUAA>(iECPFvrF*~4;(CK2mSC+CCjQ|MRtII_h%iItt5n(zK+#aG-=otFbWo^eZuhiw8qQPdUBdpE zP)B;zmLS8kr#25Ki}+R7hp3Kj26D1Be?0;RwvW=(VG#qG_CH=^fLuX!fY=2=<8U6TFW zJ0Z$1#={v(pgKPakj=7V@4ZBE#6_XC*XGO!k$?Ut{C&#fo?kMv)gw~G7ye6N5af*6 zpcH*}6VXZsDA52sN#)ET)42U&6{u^05pU~o3_ZCpJL14$p~m^60KoIT-_ z`6=v=#P^H~VpWn8w3chx#kF*xQ5xPF3-4@u`_xy)#S`gxs73ndvF!`}qDYaJn(fD< zSbZ4G7zF4s7|@~p#6TKc0*RQ>5$fXHxB*~y_>sn4zFWbwt@Jf+^BqhC@~K;dS5cY{ zq#$_kX5C4k2vHs{mJr4a(7KQ-Nlb!+2KLy$`0{Or2fN=75iX>0`gA+7v^=U{CI~I% zay<0V`G|zZe#eM{R1*}aNn*Nr3?SLwaEUL2IY}Sk=Q=BFgK3OQUrl4UgP4Rl@$b~0l}aNsg)^SPmLt0DoqlQXe4eF-^=5Y6M0K>xT;)-)e& ziIJ-CIKN;b^&rc{XvS@XiX$i{88^xUPg&TQ511LIVJ)khctN+Tts_XHn)MR?~W`U+>Ciy`?y$*gsJwOZ7}P7bvHJ61ETSeWJ5Mx zB~?aePob>H5#BMUpm|kbOF))Yr@Nt*p$O3>*5rtf}j%CC^EE{m)X+E85=v z+bi<(A-spnV?F9KOx}$m`l<~`_%}1+zw)QMU<+xck7`rp@?aT+!lH7hSGZ=aR@aBJ_LDz$ z2l&~O&?w{bai3~hMJ@ZBJcqCuYJC*OK_Q>^?vyLh+nJJG!sn8^B;s)?Kqj+mYnBy8 z)`Ezw&=jC_H+zEUx@z2bG2-KJyB^4RK$5mRl+y*M~oTo>FPQlB%6^klRNf z;&+O@O!FnA3zKMLumnh7>Kp*977>QWTHN*|o85x2r*|h~u!x#>seH;ZhkT6TXev7e zsq_nOdt{Mv8k|sb{oHJ#if#nLmpsXl)rLITLyt5u8ae%-H@v0;gyp`mF|iSWglP6; zov%{xNI1XQFefs`?%JccQpIsVf1-AM4KaukHO}qgIIveb39+yLHB-5S+`Hr}VEqK? zhQv1Tg*G>>M3Aa231M%lL?UAZl!svHfc*4F!^q&cL42`m;TeN7)~s7g@>Yx=3d(HQ zBX0VzMn^0Dp|B0fiDwvJDT(=mObh9vPueyqZRSbp7DMO>j<@FPab96{qx~S0Q+7F{FeP|&GE`uKQ zCH+tx^s6c0yq>E1eJI$biI+yBxdt$lu^Z;*M5MAjL1pQ9R$j3F<*|izFGAC58V)?w zS4MhN*~!^?_N!gIrrDxBT)l@(pDO}9K|>XFbFj#AW1P4J(W(Hsb7~BrZFxQz=1Btb zMlW6)O4qf1$-u)zSzZ|jR`e85VI|qOl0`W2TsS7nDvI?66bpafJ4g3o(AQBW)ZRht z>@dXnZ%(8f+GrZ5I9r|oKUAZzfxm!6^o~9MWA04H>W7Iih`|Tmh&cJ{xW3maPC{4g z`4ILri0E01KaB|xbSC=I0F&!49g@s^51QQ|H%U485xO=c7U80gYMPna{X3@Mt zrR+zN;y~F=dFB8bvlg63DSQ1Q!ohWLKbuM|W5ntdw1zD(=j6cg^&mpfI62%>AU**GGc{%x(fE?8yMK|7SmIF|A>R0u=! zi7JvNK&e3U7)1>*d=8;&3R*&s_Hh-POh46LuCs7Fb6g~xwGWQGUr3hjsiKnPSrT!p8_ejrql%4I}9UTUJc5Q^D@{j4*aD{ zOsGEw4DTRvsUnOlck3 z%JXc+%luTNieTpX$dqMIEU8@INVF|REz6FY3G!^=Z|SdCDYoeG>z-l=T!S$uI%qcU^7@O|nL z$`FzjS5AEc-2?aTklF2L!{8Jmzd>gAiH&_~(3i~Jw5bqi^K76mP@D=HruM!~5^{sH z9Q%Q$T!oCteSjm@-v;vSfO}{W0e8iCgfdQ+8KaC7ZBW0`q`Zb9_P?+UaH0|ZxY&z5 zEHTgSE~v_l+~B9bb1*E6P<#Y=^BG*ZQ)t}g_Lv{N%Y-AeHJ+8ZZsuKyTiva-syOEw zwdyDZ?32X3YU4|*lb-mgZ_7>{5?Gus`oXj)80LF>@)(EEg&UytS;I7uA5xJ_{t(Mp zO*CNkO8C)MxbhLGnVdpTd7k5;|L|j-q@m(_ao<%#>fL++tZP_SUv(=}Wn()2y|HW7 z3(Gg#jkvNEP=G80m<>yk%l|{edC>Ua&UNUDeaHh84 zBrJ;b!&%RQ8Ag8wiIDvk#p)AZJ;D^DHbR@x$M3JU9I*R`@8HQ!;qc&oY7 zh+bGl7wriqrXGiknT%P^T9f6SH?J(3{Jwq$m$AI?eAB3EvVeVa+Os1oOid7wrZO}R zwfx^p{L#G~6ysNP2gB(eD&6Bmi>0|C`DJ&67;$NYLYZpKR8AT`;S@9bL25Z8I0Ysm zZgz+FL>8^;8Q#t(SPdPIj7kW6u*=&g_H z4dhSO)o@*pm((Bj;gT<^OjPV*Ks{SEggt{H;-V?!1b5o7?d$$GE^o!D0jj6V4i2pU zU|`y$KUWIEIWY7$3IxQtTCZMeK%(EMzj~LvfI^G9_Mn@Jv+4~C{@N!+y8hzHkwXxR zP)C9A{2S4&>L8m`ge~3TFctkv{x_h%1jc4)qvfA`dFI^lNy>>sRhrYA&7aGjoS8Yw zLXBOXs5YdyR z!0UPR;s02^C)dSyDlGVjouLScCDePYP4NYbak>)85a+?fFvgM; z$BNA$7g4)k?-<5s_!W<$^hL}Qo|m~joeqH9{uN(IV1$^RrJY}>VAM9PzJLLe1_^z( z{uWKD5^!aH`w0Gt+Pt!h{#SC^0nh#uYXYiz28 zX}Ol<*F{I|dH3K0%0|68Nxv#!lV#cdHhj6?c(m%s+IS%53MP@Ar`*>kd@xOEk0a7z zs`nDLixQ_>#K|Ng0}W|VUO4plJuhm|TGimAF+xH{{6e(;;PLzn%f3&8p2laAj|5cN zjuu81(F%Xn7=1|Ox~AJ$WsY??tvCj4t9rQZ2{#Et&Uocz)5A9{#r<@1CAxCmaABs1 zCZfrusiet^i#MvQr%Z^nXcF0^1%e9G4efl5i|Xy#h6Rxj1O||hh9Q&&rBjA35d zRa(01y~g`~j_-NDf8P84{l_r}GkfoAuWMZ^&UK!h5!?iDrfRP5b7p*Og8aoXn%H<% z#KAJ;i(kI)ISIp}m-Y0Ml%l#8$u4<#@Ck0#tn$ITOh3LP9rGM^EL3^mfx^5!d})7A z@~^HJP1P2SQ!BFkLz<-YdGfzvp2-O<>0=wMm@{-V~Ib(3iMCwaQb|YMoBh zrE{eKR3<+Eq^Dx%67cb$J5o8jg^OYO%mHDlr%w@Js>Yxtj9nSECq0QgxIW=pqYc=! zV6=d57;l#y48^tERY+>b`F_W##O}K8!(Mstj5+-F&WeRrgSMy!4v#0+rcrnPAC);n z%kKxQGer46c#@l_4Pncn)$1~?tE3M@1Uf3>FhIO_Lo9tz5y&E`&{W?+&nK7%d4o| zLMxphD}Ha*yl0y~PLA!|h2N?@!2V;3OaUB8d3n{pBNTIMrh03Lf9=%QC%USC&EXuB zw((Cd$8A7OyYy1e4c3yr?JMXsk9x_0lyPEkfgNnHf`I4z-uQA#WqP8}G3l^(N{AX7sd3m+#c&tg<%0cIK@*?Y z0;sO(&$K5F6Q2Bx*Qj;-daDWhlDRKp0!3Gr!kuJMKp;efhW>IHao{uA%ydO_hM!6c z?yUC>VX~o)X7+zJEs1o*iIkn~d(1iU74;$ZH&8D(_-#LKPfegfEW31o)1J|9lJS}| z>*E&kdoxf=nvZ`lHCh4AZxwQk&Z!f13k>2O>W2T1ai+wGU7*=+OXsdc?kWoHB*B1G zw^c>|oxF-O^r)%^ zHq>VsO4Ch4gi|S<6=1Kw6~I~8Sd*E^IXsh`Y0-4h_vG49jCV!hMbpUs=S)sg-#_mB zsbN3p<3(V#i5a)#Y6jm~aDMlxVh{u`v8K633?FtpQ>l7-bG5r-*pav8!)If1`r>)? zu;vr&o=ufu4Zdy12Fs&Y)3zE(V5)UaEojUu<`&d$VULiux=A@`BrCqgG?YoY*YD{u zGwc=oZF}iCbGA9?y?9^{_13n_{Qn7NHX5DYtsNnLvdBisJqbBqcFjLz{^$pgf6gKF zexI2gKI6HvoD)L$mOCTutOJQ?r^cV+|v5Y z{|IyMVt?p-t*iUTCm^#~o6qfF$oH?gB%^;*?8W{{OGP0d z+cJgY;6D=QLL|1yn=C?oCJCWRPPs?LXh1YgDbUp;Q_p=q;v&$V+DC)y8zi7d%Q>cD zN#ioj5F9NWC7>#qrnaUA6PRKRL=vpHsqxds4)D1!?8NrMORLxJez6o}-!5?JC$M|j z7KA5hHZ#bg#p2Ub2%BlG-&!_|w~v|C2$4-43v2k`1gnP2XYpCd67Y9sD9JoR#$bt~h} z0TQOZ8sEnMigoeV2Sy~a!7ZYzPYIw=<9Sl z@!L&uW{#u~OsnEVX5EBGL83N?r{Clbq2lg5tquE?YN*R&dpGX+8eF%^=Br5J=JWo* zH&2Z7@z`cqs#s$#S^E(-Ucc;K?B8?5fuD(yp0&d}`UANe=l*JZRtNZpKCO*|-;?Md6csP<2L+Z1Onj*BEJBAve9(z#MG zduUV#<@ee!|F!~;@*0UN)kMv#pecMP@1t*RsQU0vM=qC7hnpi76^`*D-TZI)T}b>@ z#Odw7!+(}=STwV__D!Ovs#OFrQi~!RatD}Fk&%zg**-Lx_Y3`tbp~=@>@QBJp%(4U z0F7FHH7iF0oAt-$Ub1})}a7KEXaTH?QW&!I zr#eCoSD3d_eWlcs$9ViKv;a?TIm{rJ?;-9Eh9n z=sh}kZPt3njN5sR=8_?LwOA-yfS#SHaaRq+ap@hGL1SOg9dJMyD0x_~j7)w{PxKesuzyBYHJOxN5pC#hmJU zt;>jLC2{r4Cp%bur0{6mPv5Rqdz&OI()MNr$G8lSBLgSuv>W7CjgW`pROuP# zAzX*qrP+geooWhV!MqM4LxO)egltyUd2RRTOclS~5A8J4iZN`FxH%5W zMduh)Ne@N5sb(zwz zKwi7NHhN8WMLs>$Ajhj6fB$ozkA}~!13<*~1{aEpE(=OByfr#tv0<|*KT+rV^X+&I z05z|qjs|beMP+Zc%f>%}t)&k$9uUfWB$Ls{m8*Y!>uVrGAkzai>0DggKn7ezO~o5> zOcDpgVOi`f#L`|^vFkX$#fDZ5kN(WM{K@cA$v8e%85A+Fw91@q5ZfR0pR3aiP@@Wf zS8c@vt~V%RlB}9zCax|xzv7?Ba%`RV*1Gfi`+ggk><%v3*1ykx6{T6uxg7G_{+9fhI~u_>ti3%axwetpJ!Btl{(d#jun(}{ zHU6#{N}i=ShD`YbYUL58~P@L(G*(-AbR*_#)8O)v?mr+0STfErwdZ|h!p2tIn`S)^DV&s!qhpygeOO)Jc z<%!W`xx7*6Tnek?KKG)pQuP)YlGRx9(gd3hvJc5)p>n9Z^I)U`y{|vUQ)B?St?%T_ znCi9~&i(H=mV{A)HT>LPj{%)Zv}Pjd{5ZM!64q9ZE%v-JYP4IVrYPmL`sBT<39k^r zHiXysMuVEGV}d60)}>B0-uk71%1RHL!?Uvcq}OAYZ!|2ve=^;5WTkd^)L=ZlZIQud z(f3LWPtl|o`ldU>-J4`W7GwTqytHRsvZs0?)x3k87`(UfaYhZ+84m{Q$#7X`NDqvY zRcoSB|FP^*>O-5SX=!TMY2)OUn1Ygwfvga|itZi%MX~IrKG;mQpums-EB4~K#ba-Y zNrNWH3x4M`_L!bbxziK*9^25UtoOSF`RN+f9;mRio6S$MBS|snMwqjo`1t9y8%NxT zJEW1K(p>E!YlZQuVQTSX+=`=ZKJyHbUB9R;zLzffqG7Q870VOY4F#w=l<8MOXj4y5u)l(BVwHVL!%u6!+UGcgz-4;=xC}#JJ-#FYMgU zV^`SYJ%|;4Gv7Jvj0ff_VaW)zWPUutsjIT!Gf%MsQscrLcA+sE8yH80OOsl0uIO3ruuhaBj}4h4MxAj+tqQhtI)XQP zT;&oBj$SC9pWVj*YyOFNfp+kmFHfb+?Ztq2adD!arN|Z+_6<*+V{#iYgBJo@+4drb zer)&f9qyn64^Qq?Z7IWYJf(bUyx`^dT_qt9izRRPFwzaO)mquI6=;)>so@ppuxP?%oIW_6h4TRosGK}8k&jT(q~~e0%Ue>GL)_8Jv;K9(mg!< zOcK+&z$y)shsD7ZN(#7sOlfD7!u$(pHK35gFE9!3*U_qY*zQF{nIQn0d z3T}l1bK}s{OZv|q@oKElk4vWg_k210iLk9S%!M5pSPW0an^m_VHv`y|DP`@ z`c4W%T7UM$xAS$;kPeFP_wN8DD%Y92Uw?vb8MV#}&tEV5E*UZK5}3hBLsO9C@W`wK z!RvLfnv|%kAp^?eYLt&`wIv!pm;P7J4jpF1BxO~_1!*2WP+C=?6tsU%Df-%4wPp50 zyEi2sfhgu}sF;sZS2R6+`>dO27<7cW3%mo_9k-JNiM0RjT4>|SS_a19M&pT#fZ~OV zS@x&Y7R<;~egScHa88t{vNN3C(2fYb@BUv__<#R{uPiu-=berV|6?ouU*E5e1r{XO zR%89&UiI$|4F+FsUi_oU@}KOT|8-ULpVrL4TRH^Ujr^0_KDYzE?Atz@qx|2k)!%Ck z%Yzp?gFBV|Pi}ja8hrUYxIrQDKXe)Y`?0ekfd$k;%H+^T_J8Vg{^tjVh0wm`j(hZj z|Jp==W1_|bdzRTSJn5g@wj zhq^$Jn$YY+8o;$+3UoajHYzv`|NC_vHb(o9okVGxFJO|U)`$`*#T66>nlsO!NSx*W z-Z&`}(1+N#(U2FO(Fi#rX&R`E?wyIrY&p8an`a+VnW+ zejNsj^#K|1n1<9pI4j;JfuAEZMiu41|2P8yDr7&l+>8BRpLvv)wfs{0Z{Bg16!B+J z$!`Ob2M|zSE^`91wnrfk^l(5>K>-ZT8)JNZBAuMm?aagT1)vU=Y6qk#$m1-qQO0LbAI_5*W~cq3+;aHKo+P3{Vq;;;oUKxMwK}Z^*aFrWZQDI z6kX<5Fm&yYIW*6b^qh%+TCilwcNV=w(*BG)p#S@*airIpNEQ|CCs<%5rNzvMzbTj$*oS&qhJ2o9+>fN-}Xvs6{#{6*3O2y}Z zZf3dVy_k;9D_VNbpRpb<$9=!vqKMemjS?GpIHoPb#j+dyXk<%(#RBmq_{d)F8( zY7py{QM~0PHQvrXQ8!Zi9aJ7@X4p@9ZLBzJ#-Ex3X2?ONwcYe)p)KKyO`l47``vfH zXKoMh+9sA)Bi%HQeu@CPRZkZ1vjWcc0kwt_ZXh{@&F=ysPl{TwI%P(wrsK~PLz>53 z9Nh|)>Etm2M5pQ z&fP(=x2q9#W1_x|n_qqa9;8Kp9_!~0n|D^+Y`MO2cKX<*Oo84&xWs4~i&e#bM^PXL z5F|?|0_By~?WAnXU3?l%3s_P{Fpx*4COeJ$gW+x6)ttUdS!lu)dNh?Yx5ADBy)T&? zSnO)0B8{VYuZ>a_hYfl3Ip0g-akf^F^CU2!5{yf-*#n9gz%uMs_(cqJ8$gEm$!d%N zn7K-N<&l)IR+Y3aI+$H$z<^ozuggI@r8(OEFW!5In*VS|`8*b5wQ&GsT&Cc6 z@X-_q`^8I+*yMwf=bLT55WPRZh@7(;tye5F7b6>mxWT@2x;&hscvdjOUz_TNvT5K_ zk@CgVsvRjc=XWyB4zs&87E9aZyDpgW8&=WiGd)CAeWuJ+Cu z=U}RR&2ouXyxi!} zEQZ;s{fLc~At#G84GVp6SPqQ!~Zo!f2(LLfP3mnF$$|nZ}Qt3uC zE!P{y*I^DiK?R;U5*;S5-P$|^BJ>wI9P*7428jZ;0Do2ZPJz^A(MHCSp-o@Hqo7;@ zeP(Wp$E$!-s_)+>P2KYFm1Av%C_1=R4ZwQ8)hD#%UsN|FeCq|dh92f+)D8c-2z*lj zOx61ih_t**mjR7G9Y7Wb2rD6KAJq;?o(JJ`e`Z*q$fFg0(wPDRe6KIS?_EayJv!pl z@yW$}8-rW`*ViJO)Tt8L6JqB;MK57y$_V%kTrnV6*mky2p6Rx$h1o>n0p<^u?MZ0f z!Xiw6^Nc)mB3_bay*Sg<-~nx>7Pz|5 z^IA6rx_hu~DB&d#nnoh!vkz@o^ zQ3^{&&rP}>l)G~-L3FM_;|wDkWrWUv4kV`L8JdMo*hHO7=FAEBwBwu{Ui|2z#SbBL z_h{d#65JDc1-JKRI)Ysw3QfWOA_B9UW)a^LX6r_*JOsAyL}|=W1u#Q$ZELI`7{Ajq zt1GCwn*lyxxrjF>jXf&i?dm7?KZPsw)2DXN_ji zed$2#&{(B9)Z)S3{U||>rElIqb@-^Kz>~qC2sBu#VphH(@SLqC zan!9w5+Y3pPCULd;hA5z;s$s>y8f`7eB|ON=wf(`j?JFHvSOQk1g?_h&m!aw-H@OF zs)t$q#kydmjBS``3#BiA&0wJ_es!|kfNDll;w|UWqj~_vzkVsx*@Wn=xXbld&#CYd zm_Sj8EPv8r+~%uUK{U?&>?eFDvgVWpYHqbRXfFkZ#03CBXQ4YVlw(Rod-hnA zDexoSs%)y0f&6;;hdo)WvBs2~bIx6~25x6kfUV3-bUNJ)bwB@Fe6PX}zX__7!-rVRnp+58^6(8F{&Ovcf?( z@{^?FqPW)xDYLp;X6}a*a*ps)zGkST61aSbj)^VKkteRG49s-IsyH|Ms*vR;Ye6O!A?1Me_K|C&1!mq1*Kf-J@a7 zX}qxrTv8MyPDwMW^-Z(wB(W|@BFW-c5bDs+yW2=CDbh3&pqTvzj_^_*emgqo(-8TU zKo?u}S?G(x7mBq$sOe1$s~oyTCppE~lJrYr7&k88=OK>*X0Hz7?F^?jyj`&symt7( zw^bs(QzrlGM=nNBPirK_rX7`!;UJ2s8T{&s_6#He@@f#tio= zH>$I68J;CLB)Da~QFSjz|2)@q!Ut`| z?5S?d$F~B=t%+P-)UZzHp8b^4n&GIvxP*&14^3} zMXBVvlA8xzQdiE=iVse=#vEx?wCzvssG0D8T zQ~p*Q2+k>2e^DtJM3aTVg!-vR$V11aXX35gY9&&!*%c2P0Y3S+d!$Vv(nJ>Sk}=)$ z-ZNO}Y)-LDeLJPx@37c(t7b|)-1}0MNrEPypp~bhgQV$?tQQ^XM`JND3zJ*F0mR9m**rn@P^hP^IktlwgB+*~Pqn5m9_JQxC&Q01h4H zEnZw9CB6iqvxy&$0OD?$rZjZsb|lL#Rne`1kY-T!=iuSCppU?#0AY@8qZM*X1Fl_y z+#ta(+%;x9R~xvxoD{yJi+*+l*3+NRY;ICUbiRBC4+<$9`ST(DYt_l1*mlH~9^q3J z2IQPB%%P}jB{(L2YjcA;qc$80{ z`#MgJH2)pZXVk(N4+Zd?4*2X+kXi4syqMp#yMnMZJbUTFe!ZG!Nr|TZyU0qKmP?W{ zs-R8-JDEzGwZ^bkQ(>GDTZV76nQQ>T$aUwjLIl+TL0d8(jXQyS{Nv!3#0!@kRjQZF zXv0L@*owLkU=eAw#e+ahl{ntb)2MD9wZTPq5^fM~5^9bQR+h}Z>f_Qp=z`UNZg9EXGl8KUCO?Zg z>a_)2R!T$5PpO4ZEx7QAQ*@XTZGMHvZ|jv0Dnuj{k#{=@I?_Vccy|B)PsO3`v)HLYo=7ivW*EA z8OPjhs;#1rJJZ^cDNK$0u3=Q<@+biN_;#C7Hj?CP3L{tT|8M~$D8@Y(m|}4*JVKs~ zuM07<%L%uc&=%Q=u$2j1HhBH%vAp&xPJjE?6-nPPCEHF0;lEheF+#fErFafEPHuaP zAG2rEyX-|Kb-Z`HW8M30j=Z)0+gCHEhNXvo)(McQA`l>JK13!Uwt1;(vb=_lqaF7H4{6#<5Oz^2r7Edc3#6R1b)Mfd=XuIRHld^b*UgL0n(wK{`OqXsul~IQ(t8$h(_9V zC3utqw# zawx9AXU&ZX!HL}o0KDkoSJQ$%n+~Xf^g0;Sm>IlMzlX@9Rq+Bos3~!ZIcJ-|TcouhsR>oml1s`qWq9-RUjA7$D9Okd4&m!pSu9|&jFP?G zGMulRRFn#;3{BB!LWm~0e`4}myr&)R*Lww9`KC-FiN|JhUCFH=&^lW)5=&rAAM8Ir2p%|y% z3awiGL8Aq;3NttSsk!4MZjQDkm$8k&7E|vvh}X&J{7x3pRHu?);DK&81_*qG)IOmy zm(IYj&i;h1AT0u{ik%{ZUXrG*m6i~SJmN5Smk#^Ku0?XG_sD5tlTi4PD+p^D0_5=? zqLO(;UsPQsYg;n=c2<{z2DkJ%-oi2SdvQOCY|mx3KI(Wc>+QuE=DY4Xg;S}+#G1^@ z8kAjGz9}d})v&KtNxeBnYSfW}EKF@BnnFH-o8kTfVzJ1rl#@Vqpc?I(W*#L{*>HR^ zXV7Fb#)Pewbg06p5A79?9oz53`c+zsdT{oAZ-Y2^DY+3a3NB;)Dv^pK01A7MShF4+ zqF}F(PwN{np^=LprS-!g$ZXitv>s~ z<>&-ayqTAFH~nKaLhK;|T>|<3sYl+;CIKtfJq1H`87czZxR0+1jS{}cVI%17cwSH) zhPbb!X-uY+U&n9AgZA=tm=;0YHd##cYC_&LmbtO0L~A?(nDyUekt4v8_F!v$n^z#E zPz@G~0u_ek{yENevS%TZbcDjAqfpt`tpsM4c>Y^<#^GxnWA!>x?R||x!h-5u52vk% zlnSV4uYO1u^;j9$nZGVE@h`wG#Q+B6McBwRIUIMcC38Hhj+^{4T8km%04{D9sY(E?SOu z4EY^ZmDNHtk)Hk)cOWmVZ{DT4Sze7%oF_cAJqD~^{@+%wpo(}dRP*QycQu=cKa(>l zU&xVxUnBE`0~puf;}Rl;B5L10Pp~eDgvD^STZ3}IE^0An4iGh&J!C-9t7 zk1C2LALt#;sm`4LjW6$F-P`J;pl#5*#i63oF4avXjOa>* zb~f1q7oQ+RMwO0=bvM?l3tQF3{L)5{wkyovy(B`loZwHhMz$~SPlNhe4sm>~gsVPwIvm|Q*#Hx~1 zz~*D#?52nt2|Xk4vGTP+$z9Cn*FNpZSN0grWogAK4(AleZ{2ulM=Z+kW?6*htWNLo z;^w2j!ta^j&^Spy2Q=LK#VXg9ChrfvCp_blds5n_=-cH-nQ1XrmlvR5*|?*D#`_Gv z#I)J@pcCdG>h=12Wh~9!p_z0Dm1F@MLob2iNWUAzI^t@cKOTv+( z+n$y(-k`k;>gSG(b^vMYJY^tkJ(h{{ZeqG|5>}`xjOh{lQ3d1R3U4l4Gn0_Q;g;8DiQ%vLzfz{!c(>m@1o53c?!1sZMvLti~WxA z*R$4AP(1%+Jxk2b?-6ECKP-}ynqrmNCS!O+~9 znkvd#swGiL-mNN4e#^0?s-Q?i~1L!dc22Ad>L)^V4HN#4i?%e$`xZj zxzZ^{YD~M0H&+LN?_AHAIqR;G5vB_uZ3TDn= zBlHf0-vy9?7{6qRBp&)%)@f{UKTYSSOcO!{e9HS#iL}0F+TrZIpC~xSoqd(72QLZg z7Oals`@@Xgq!%3<875EmFDF zwdvO6TfQ>4S;{$u-rVdeStHFpnx_M2#_UxilIEJehM=Za9eo1{qhw~{z}uQFfpF1} zFv@#-0d*?G{FwhlP%yehpqnq}BU}_wO+Q45YQ-gh7>~Na;~;Ia7opX8h{6sC{yeJ{BKzQX<< z6PA84cD%Qtcq0AnH0>1?pm^w()89w%PEsd9v}~sHDyH=smpBwE2Q|0sArR<)(w+Yd zQ_1oeu57HvgToW9TuB_dpEL5D`lXKAazsP8F@Jix&$%O%vhsuR6sH+ za6nnD$uNy;dK?R?h}7tVp>%!Q@Kq0sk>(}|GmRy|o2g1rjP?U9qA1Paz4O{Lsu#bT zWN58tm|!sBt`3&$kEA^zdcSfpyDCB#SDAfXDieMq=Y@|Dja8sUA?JLns7%TK?akO8 zH(_F_-B68Oqhde))^=4J?5TZ-;Jz4IqnK~c#7)E(Gc8B%?e2XP1XDTa&a|Y|+jGQr z%CE@SJ+?f(#)DB6b7;E`Iv)vH5SPDDk_b&3`ttJ*jvy(mh}3h==98sa!4J_I#`Ntt zZ+NnVidX94eGkSQ+o&#pOhM@|n~Wapq1xcmrfUyf!Q^pvOFKU(lJve#2WY@wkV5jv z@-fUFos%Cs&ijM1t0{H-v3Cr1{}t{;&L4&ihpj7%OrXq4R5}}y3PtqVyOdz-*)qN3 z+}X43g41|(3Ags4to`{bgGrf}bL`B$?8_Vu)kZo;^{vR zEp%l$ObHKqhiikzw`UzQ!uaXu@g z{_CjT-5B~Rf3_$0O>o2Crh4ePHbpY{e2YbT2BFG#61g)U%f~|13_BA(X-$D89;R0S zaI>j?^EYyXgPjkR^jolwsOs>l#~o;a08=1O=FFt>9xK8rhCQMB%Wf0NY-}jEjYRw; z60Y)_+UKnWt3|;X-)U0y_nsZ1QC1`Gt(u05OdF%KARNPCEQU$dtH(aJx!6{tRIxqb z_Tcsjjsa7i|2|Q+m5h_`sj)OxWO=d+sa{jESK@y4aUe@*@`4D>r42|-#&D}4W(4H z7%uJ4;`$ZO9WuBA%FWZCK$M#-kvoH4BI__|st+o~-yW$w`=0){prn+gha(ZN4l0pf?_ zS_QLa_ulJ-{Qwcr=vO4j>OX*yy=v~z$Aui~4P<(d6{sOG^HQ#%ZBgDR_b#b3bflKO zLH}XiCSfj44oxS>8w%(V#CiEI6Lsd zUE=+C-X78<+l6EWJe=;)2i`ZIK9Drk`~B)OVW)jR{B;7NGmhMo#W|=0k?Q9_~y`*gCd`EK|nkT9n~EXf~K zH9F6RF{w_TAE)w6&7&hikC#lsE6Cm|q0}U=EIg=ZK9jmRLo9{@HYxI;uXPm+mlMx$ zoqVYrG1zmL=Ex{+YNFrkQvG*Wl()Ro!@xaK$%oAYNf;&kc}98}jK`v9PR(k%th*v{ z`Hshjk#Lh_E0}$vI=qm=ZB&U3-PUDq@Q&V&GNSp2$4Xt2* zyP%3v+%iJhgwVWFv@id~=Jk*X1V`)+vb~N$6>{=HPE2-a!o`uPz8hY#n(n9rIh~_9 zPSlhI`f=Vf4l5cruX+re7{6Z=AfNaKK9%;yW8lqgQcie%!Fm52cZjqjuG(O?oHsb5 z&h~LT*+}f_TyAr2A8U^MgJ~t7lt{}SIK7V^JPY~>5?`UKXOYMYdhE|Rx43IF zLEq7wntPs{OVq8#D~F?%Ek*4zeV+aUdwc+|PaA@hkH5i5GNK3Fre zVPW?5_*izGzy}l(?uCJT+6Sf6`zlP$?RLzFDTp~nNSIG*T zAY<$ZZk*<%x~DX`ZsmbImKSv^)xfTTsI>CWxLHZl$jFr{ZaO_y3|Ni}Rhx*=juti! z)z{--F|WDQi&}jvH!SN{KvN!;{kTTZJD7R@O&$>>0LE<<>x&!e+ZDz336|~%3_xwlp{TEWB5;^gPzWn zPH*%$Jg&HRXq-a*3fX0l3)6cZ2eLurv1EK;-}9!SiGxkYj&dEHph;_klF8GyZNUGJ zk1m`1h&bK^DZ;!%GFzq(KOuB~aPZ!eKDLZafl19pUWsF{rtL<;ZeqTJh~=bhO?b*? z`*Y0B(-=BFM*=2h@^2^9O7V6(v-~&I!n{yLdd0d7dclyX2pLya-GmBszPSlf_<#}x zx!^Zh@995pHtA`XfxsVxIpNYZ%|0W$_YPYgj)8O_1f32H61TPlJ%Q|+jv|O_c=JN@ zjZ%jc9}k2D1y_HeH19Hg*12lTat&s8p7M~UcxGhXm!=8ji8`24efi=u$-t|nj?i8D zW)ar%#kV=>?m*%*Sp;s{WK%Vk^et2IrzdMA1(kjUPI0JAy4cIs61zo4s&{Qx}SEZy+lrMR$FU{IJ@{>4{hzbYjF zt;a|g;3h!ocYerG#!<||H&R9;_{K`vN1y&XQ(RTSj1*nncYUeg&z5?Wtk(&Hc2T}D z2^JU>$e1}O+LMP%o#`E5P?=bE3jEvXn_o`&(fjOtt{CIQB9z^~)#J=DFLC(Us(*P9fH@R$3&_7`gitlDTCoBU(R#t|vq{WwmcV?pSX`ARhPoMe*SvhiEcDwgCDU zxh@kuD*1uxZy#xe8qzTIve|#n;M-7RyCGu+&MB_hj` z{SrrMol%D{hGqvJcJ6NJ??^$#_n$(y$>?osOJUH?i_5N`k>LCil;^Vugmg-a9h=1_ zf~q?X)4<#ncL)(Z9pYh6P(G9SjV9F33;ttEO(gfflRP=!(`<9WC4ZY*5E&7nU06Ef z8}_)8?B^u1{7PY*{3qSkGh}+P(s2=-Hi7(-1-9|QTAr$kJyGS%F}Z}Qf>v?lqyua2 ze30<|Xw}U&t6n&DP&z^w;Z^?Z)1Y%EhpB}N`c$;d!=52i zCgh4bvf-~Xc)K6Ia`ySDmH%vJ=$Dot1~v=w@C~1}ayzSIeT+MVD3 z@`GlZRgGfe>1E2Anrw*P9ziqtz?+`!c)u}9^DHFqvYf!TF*iRn=*1c*fd4T?+aB`{ zL)@*QBBo_zk3e-2qS9LGU+(>3fIQ8?;dlk|icq}*%HJ;1C2+SdpH_S&&Bcr0s`5_M zqio#W9uqlb^Xb2C9sBUUx#}CU8?;@_QL3Cq^!!A;=MCLY59H3U4({xAK4Q4X@K^rsot^gMzKV!NRI?Tf)=Ax>a$32uUKd=?yKQxLh-mAwa)x+W>*ijuN&k$$*W zxHN6KN3bkqTIxq>g9mR{rSh({n@cRqRi^z&W3{y&u#MKSRX8m)qhPmfs#YuXeq;`k)v2~Na2vKvQX3-A$puKRUVWDLqrv4cxMQ7n}@b z4WlgYqUdTq;RzLu@&`}@SmY0zg7h{ZWMZe&Rl$k%7MGdoJUjaF_1hcD7QTpyb+zG7 zXek91A1b1h^KH;$CW4i9AnG$;6R8F*32pr@SrBOOb_9KkT2v6fZ{D zx4t1Ck=c{CihQjNo3p4*{R?_&At{`aG2Zsn-q_xW$Sn6u;!bw(z-+dWQHGf?W}21vL^( z!afsGuym%78&@h)m@@Xiq+7FOr!d8UyfukNHnv|`A%Ylk+iN_(wFH<7OT5| z+jh-)w02D(k5lep_ZMD#ajIA@EtnFMBf6Ap^K6Q(>YI|59@kULs>_&!!8kvH5$1i( z74$_8IBDrDJ9n<}I!awnIv!!7vH2YZ_uM!j5lIRE@CHzc<&m zBNf)bwj>mv?Fiaiqaj5)j=iR%0)F}jH|09{c0Vxo!xE4WVdKbpt8Zzim*H4()t;vz z+p#W#Gw9a6v=}w>xmuHcu#e|eQtKO*CpJ#1Wdw?3c}GMS_rQZb|&dY*!R3qGqCrSpFSDeNZ>Jyy4Hb21&^D8gDk1Hi~4CtpW= z4jbF0t3>&(pUSeObpy+1KnBs9e`Y^=`%wQIQsm31bXZr@R_)CC?CjRxtWzzE#8XJ} z?Sr@a&-%PT^PLrFn+-aPWh8nQziwO-o!bU8Qr+c8Xb^vZpBzKYeGUkQ!HaVJ%k(e? zrdldb9i6AOWOQOR3e)*ZG#jXVe&LZC#~ML%w|V9D25k1+{a|M34p6EE-;6R7rw5I;R>Gt|*?FElbw2s#%c0)VlNfMN z{rNZBCH}fTl|9doC_!(z&-dzzge~#wdU#4;ooR-lKdFw3U{k&28HUas>SYME2w<$OyJwoyUqXwX*_T4TXji%TXn?<^74SaWiv30*Zq|o8O_EQfH&1BQ?=&!Q4_ln z$5lAzrzK}dvY{OsBZ3qq*OM#cL$;LiPxj9{BKQ7PlS-17Tcyu5P5k<8b1Ar4H~acq zrN_|jpCZd5U+watQWA(rQnYbN+;rZj6Xl@x&jV;|LfK)dSJ~vb-|9!{-ztA25s#tv zmN<3aS1kNGzq{%EOI14%Gulk={2M=Y)+zUH_arKYJ7XA^2?ui@F)O(cJo zU*`rn&#lFDP(#Uh7OT>PuBcF#Pd25^ZleKeKSJCLxr&oDSx#g*lLhT+XVWFHR}5>p z-p}%!H}y9mXXpN+@X*83AOx?_I?s-{W@*A%2it>o44r(=lVsh|My8R}@D`WW>R;q? z>GiF?%`z{-np&@&2DMH4WL-=qwK7%xKegPK(vDA&pX2=8e04^(!3sXlFoe*lm#LN@ z4V8)|zK~!v+DRBc(EMtCnpl{;TSa0~F>^NOr(G=M;5BFkm_Ld?wtUQGik{KNik&6Y zLYmCh@Nn)6f(~9!V(5Z1^+i;Vb>^#nD$5j8e&5o@MGXsy>#&Q6)2&OcfUnA}xN(UE z(rjEsEn|EfLON%tB+#Ye@h2(V#`Enb&>%YodW|MO;Glepd$-`Zp9{devj!;3G7#6( zo3kp&m5kL$98ZDsC;GN9jdKUYs48#_I}zi1$}5X_H9-upu~WOvLxY(t33LP{6b1k<6`G-pv7Gue>1(S==cr#;PYkEO!*B^ z?oW^mcOMLPRvYwdOJivqF)x{i)lbyOB=Yq4Hv{p3CH?G)crUhzM05VWv}Vq2>1+uS zNM*M${nOP?TVUpuax9&6EBx3}&@yttbHG#wnQy7qEX0R+ z+17ZzH`OjLe`v7nH6xdMQH)?Vf9;H#PsErnbE9+E>y>s{pnVZdbOdy7!>-J-$svrD z?%X2-Qsc|WlRiD+rt9s4aDn`E&!r|>G__X#Mls?d{eq^&xbARPk zvd#h3eizyg54&US6fe3f$zzl83+YtZ#!~i(%I3=Q*T)3jn|~S>rI;FSuHBJbksiqt zb)`5tOcGOsy?ipxYPvkzwhn3Zolv6 zy6XG;z8r0SbH!O>}IXDYv|2w3@zhS zEaVbgZ3eF#nTg|4tNXr`sNy4Q#e2_8RWS-xmDFJ5+$_9|{;o#-Al-vMu5(^t#Gn%^ z%8e-ghMuPHGwLwk|57QQjv*=aMLRn7L-Gku-9n^2se0sAqLL|zZX>e?hEDne@4j*B zVTQqs){d7|ynK zN@IqU!H|RPsknsPSIcZI9t*3qTXts0X_H??*}nq!tlq;k8P+;$UX_L8mNEY$ZTKt_`MrcN9b%MeM}!4WNq{C zw3~Vkg|zELx6y@(Tj+22p;+g%+dTL`u4b{dY;6}*{xo+51J4sTXw=s6bv=5vmr90% zeCBeJaZFVXI)k-dH9ssi+D2ejb0eaxXukWw(Na{)C9v#NIO74!$#VC)d z=1!IO+OE&={w$9URNY)yU+CDh2be@)_+6E*2*|RPAnx}@v=Voa_hVdNG<{X~yFsr) z_Sok}09NTo{|l_LP2N})cEjHpD(+t#K6Qs8gN^pwLfz;HU4CT4(Ii3~ByV|PZhri-k zQ+);6Mc^k3tSa`M3GHpwtRhDLGE!|bPnjn zdR;GmjegHVDJVw$cF&i&p_-A4dLD9>n)facQ#NjoNmqt{c(Ej1tAvrzWeRR)?H5gn z`sm!+DCm%h9}`mPH1{A=t`iSUpx`}kG{V%&T*GfcdBfc%&HWsIoK^3o5=bp^1Qw8V zRJ{X%1tXkGVYlC)g=p%#5E0q;yXNvWo*o_T%@I}h45J#z#pR2?7j+~3hZy0M&&>u? zLghl2t^aV4p(CMdz{Km_a z&&SJ8Y%_gVb&A_pXcQ`O6_PShH|l!d6P{^?1}q@83rUGx?GyDbhwh^UsZqlhTl)00 z73i}htdcu7qGb|@)-0EpnSyyreG2nr#8T6uZ#YAJ?CCqJYnCn?%UQEcinVvvQzsQX zasK6E&B8v<{K>HCU-1W-(HWtH!O)pU*)qx`6Bx4IU$7 z$q^lnzW(qrM<=0EAU~&j{8qH)8$#fjeOrXYlqe!uOwreu0xWb6V#h z9~n;Ky#WbYi;`)`2%k;AhTi)=DKV(`vis)gJ0XvQ%ly#>;$m@>GUY}RQ7jOQ-H_}P zw%D~v_u|p-cvh#~r_O|FgSJ)TJ54dyY7gXW8K`pfQgZDpvSZ{lu?{+@`sQw>=8RXJ z^6eMiGcDH9BNa&6Ap!;ENUgWd*6nt{wpd9)s@6Bs_e^8zYU~8WQf1{%lsxu$4L-i2 zT}QTz*gDcNb|Fn{BI%hV$tjG5McmMC^);2vIQ*2C8D{hAC9WD&#h`}wWTr)q@MP{6 zl!zT8Evfjah)mHwf|h8_L6a8GeIvWrNGjj`9|9jIT2^sp9?Mk2-G9~lQZ<8!2s